<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e89315</article-id><article-id pub-id-type="doi">10.2196/89315</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Effectiveness of ChatGPT and DeepSeek in Urology Medical Education: Randomized Controlled Trial</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Yang</surname><given-names>Wentong</given-names></name><degrees>MBBS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Xu</surname><given-names>Ting</given-names></name><degrees>MMS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wei</surname><given-names>Junjie</given-names></name><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zheng</surname><given-names>Wentao</given-names></name><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yan</surname><given-names>Wenbo</given-names></name><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Jingkai</given-names></name><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chu</surname><given-names>Guangdi</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Niu</surname><given-names>Haitao</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Urology, The Affiliated Hospital of Qingdao University</institution><addr-line>No. 16 Jiangsu Road</addr-line><addr-line>Qingdao</addr-line><addr-line>Shandong</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Geratology, No. 971 Hospital of the People's Liberation Army Navy</institution><addr-line>Qingdao</addr-line><addr-line>Shandong</addr-line><country>China</country></aff><aff id="aff3"><institution>Qingdao Medical College, Qingdao University</institution><addr-line>Qingdao</addr-line><addr-line>Shandong</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Ghanem</surname><given-names>Omar</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Fiorentino</surname><given-names>Vincenzo</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Haitao Niu, MD, PhD, Department of Urology, The Affiliated Hospital of Qingdao University, No. 16 Jiangsu Road, Qingdao, Shandong, 266003, China, 86 186 61803117, 86 0532 82912105; <email>niuht0532@126.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e89315</elocation-id><history><date date-type="received"><day>22</day><month>12</month><year>2025</year></date><date date-type="rev-recd"><day>10</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>18</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Wentong Yang, Ting Xu, Junjie Wei, Wentao Zheng, Wenbo Yan, Jingkai Wang, Guangdi Chu, Haitao Niu. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 21.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e89315"/><abstract><sec><title>Background</title><p>Since its release in November 2022, generative AI (GenAI) tools, including ChatGPT, have gained widespread attention across various sectors, including medical education.</p></sec><sec><title>Objective</title><p>This study seeks to examine the effectiveness and feasibility of GenAI tools (ChatGPT o3&#x2011;mini [OpenAI] and DeepSeek R1) in enhancing urology teaching outcomes for medical undergraduates.</p></sec><sec sec-type="methods"><title>Methods</title><p>We assessed the accuracy of responses from ChatGPT o3-mini and DeepSeek R1 to authoritative urology multiple-choice questions. Then, a randomized controlled trial was performed to compare the learning outcomes of students using ChatGPT o3-mini and DeepSeek R1 with those using traditional learning methods. Additionally, a questionnaire was designed to survey medical undergraduates&#x2019; perspectives on the application of AI in urology education.</p></sec><sec sec-type="results"><title>Results</title><p>DeepSeek R1 demonstrated higher accuracy than ChatGPT o3-mini in answering urology-related multiple-choice questions. In the test following the self-study period, the DeepSeek R1 group surpassed both the control and ChatGPT o3-mini groups in total scores across various question types. Despite the superior scores in the ChatGPT o3-mini group, statistical significance was not achieved relative to the control group. Survey results revealed that most students had a positive attitude toward AI-assisted learning, believing it could effectively enhance medical education.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>DeepSeek-assisted self-study was associated with higher posttest scores than traditional internet-based learning, whereas ChatGPT showed numerically higher but nonsignificant results. These findings offer evidence-based insights into the embedding of GenAI within medical education frameworks, providing guidance for educators in developing teaching strategies and for institutions in formulating relevant policies.</p></sec><sec><title>Trial Registration</title><p>Chinese Clinical Trial Registry ChiCTR2600118749; https://www.chictr.org.cn/showproj.html?proj=302930</p></sec></abstract><kwd-group><kwd>generative AI</kwd><kwd>ChatGPT</kwd><kwd>DeepSeek</kwd><kwd>medical education</kwd><kwd>urology</kwd><kwd>randomized controlled trial</kwd><kwd>large language models</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Generative AI (GenAI), a subset of machine learning, can generate new content based on data from training sets [<xref ref-type="bibr" rid="ref1">1</xref>]. The advent of GenAI, exemplified by ChatGPT&#x2019;s (OpenAI) launch in late 2022, has sparked considerable interest in multiple fields, with medical education emerging as a key area of application [<xref ref-type="bibr" rid="ref2">2</xref>]. Undergraduate medical education, as the cornerstone of talent development, plays a critical role in building theoretical foundations and cultivating essential clinical skills. However, it faces significant challenges. The vast and complex nature of medical knowledge requires students to master extensive information within a restricted period. Traditional teaching methods, often focused on rote memorization, result in superficial understanding, making it difficult for students to apply knowledge flexibly [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. Moreover, the growing disparity in educational resources underscores the need for continuous updates to content and teaching methods to address the rapid advancements in medical technology and support students&#x2019; motivation and lifelong learning abilities [<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>As technology rapidly evolves, medical education is undergoing a profound transformation [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>], particularly in the realm of multimedia. The development of virtual reality-assisted teaching has enriched learning experiences, enabling students to construct a deeper conceptual understanding via immersive technologies like 3D virtual anatomy software and virtual simulation laboratories [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. Simultaneously, large language models (LLMs) such as ChatGPT o3-mini have been integrated into the medical education landscape as valuable aids, with potential applications in clinical decision-making [<xref ref-type="bibr" rid="ref11">11</xref>]. GenAI models, such as ChatGPT o3-mini and DeepSeek R1, can serve as powerful supplementary tools, offering immediate answers to students&#x2019; questions and facilitating the rapid acquisition of knowledge, thus enhancing learning efficiency [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref14">14</xref>]. Additionally, GenAI can simulate clinical scenarios, enabling students to practice clinical decision-making and diagnostic reasoning through interactive question-and-answer sessions, effectively translating theoretical knowledge into practical clinical competence [<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]. However, challenges persist, including concerns about the trustworthiness of AI-generated content, the potential for student overdependence on these tools, and unresolved ethical and legal issues [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>Given the ongoing transformation in medical education and the advantages and challenges of GenAI in undergraduate medical education, our study aims to assess whether GenAI tools (ChatGPT o3-mini and DeepSeek R1) can enhance the teaching of urology to medical undergraduates. We quantified the performance of both models on authoritative urology multiple-choice questions (MCQs) to assess their reliability as learning aids. To evaluate the practical impact of GenAI in medical education, a randomized controlled trial was designed to compare the learning outcomes of students using ChatGPT o3-mini and DeepSeek R1 with those using traditional learning methods. Additionally, a questionnaire was designed to survey medical undergraduates&#x2019; perspectives on the application of AI in urology education, gathering student feedback to establish an evidence base for guiding the future development and application of GenAI in medical education. The findings of this study will provide an empirical foundation for the integration of GenAI into medical curricula, offering recommendations for educators in shaping teaching strategies and for institutions in formulating relevant policies.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Model Introduction</title><p>In this study, we selected 2 widely recognized and applicable AI models, ChatGPT o3-mini and DeepSeek R1, to evaluate their effectiveness in undergraduate urology education. ChatGPT o3-mini, an OpenAI product based on the GPT framework, derives its core competency from training on massive text datasets. This training underpins its proficiency in parsing user intent with precision and synthesizing fluent, human-like language [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. The model demonstrates strong capabilities in language generation and logical reasoning, providing relevant knowledge responses and text generation services based on input queries. In the context of urology diagnosis and treatment, ChatGPT o3-mini can analyze case information, symptom descriptions, and other data to offer diagnostic suggestions, disease-related knowledge, and treatment recommendations, thereby assisting health care professionals in clinical decision-making.</p><p>DeepSeek R1, an LLM developed using deep learning techniques, was created by the Chinese company DeepSeek. Trained on vast multidomain datasets, it demonstrates excellent language comprehension and content generation capabilities. Compared to ChatGPT o3-mini, DeepSeek R1 exhibits superior accuracy and logical reasoning when addressing complex issues and specialized domain knowledge. In the context of urology diagnosis and treatment, DeepSeek R1 can analyze patient information, such as medical history, symptoms, and examination reports, to provide more precise diagnostic suggestions, differential diagnosis strategies, and personalized treatment recommendations. Its strong judgment capabilities enable it to better understand and analyze professional terminology and the complex logical relationships within medical texts.</p></sec><sec id="s2-2"><title>Accuracy Testing of Models</title><p>We used 185 urology-related MCQs from the National Medical Electronic Schoolbag software (Beijing Medical Vision World Technology Co, Ltd), which were selected from the standardized residency training examination in China, to assess the accuracy of responses generated by ChatGPT o3-mini and DeepSeek R1. All model interactions occurred on March 14, 2025. Each question was presented to both ChatGPT o3-mini and DeepSeek R1 using the default chat interface. The temperature was default for both models. To ensure reproducibility of our assessment, the prompts were used (details provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) and were written in English and Simplified Chinese. Web search was available for both models. <xref ref-type="fig" rid="figure1">Figure 1</xref> shows typical answer explanations generated by each model. These questions, drawn from authoritative sources within the standardized residency training examination, cover a wide range of urological knowledge, including fundamentals, diagnosis, and treatment, ensuring their representativeness and reliability. Each question was posed to the AI models 3 times, with the responses categorized into 4 levels: completely correct, more correct than incorrect, more incorrect than correct, and completely incorrect. If all 3 responses were correct, it was recorded as completely correct, earning 3 points; if 1 of the 3 responses was incorrect, it was recorded as more correct than incorrect, earning 2 points; if 2 of the 3 responses were incorrect, it was recorded as more incorrect than correct, earning 1 point; if all 3 responses were incorrect, it was recorded as completely incorrect, earning 0 points. These categories allowed for a quantitative assessment of the accuracy of the AI-generated responses [<xref ref-type="bibr" rid="ref22">22</xref>].</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Interfaces for AI to answer questions: (A) DeepSeek R1 answering A2 type questions; (B) DeepSeek R1 answering A1 type questions; (C) ChatGPT o3-mini answering image type questions; (D) ChatGPT o3-mini answering A2 type questions; and (E) ChatGPT o3-mini answering A1 type questions.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89315_fig01.png"/></fig><p>In the comparative analysis of ChatGPT o3-mini and DeepSeek R1, the 3 responses generated by each model for every question were aggregated, with a score of 3 considered a correct answer to assess the stability of models. We then calculated the overall accuracy and average scores for both models across all questions under identical conditions to assess differences in performance. Additionally, accuracy rates and average scores were calculated for various question categories (such as anatomy, diagnosis, treatment, etc) to compare the models&#x2019; performance across domains, providing a comprehensive assessment of their accuracy and strengths in urology-related medical knowledge.</p></sec><sec id="s2-3"><title>Sample Size Estimation, Participant Characteristics, and Control Group Design</title><p>This randomized controlled trial was conducted in accordance with the CONSORT-EHEALTH (Consolidated Standards of Reporting Trials of Electronic and Mobile Health Applications and Online Telehealth) checklist and guideline (<xref ref-type="supplementary-material" rid="app3">Checklist 1</xref>). For the sample size calculation comparing the means among 3 independent groups, we set the statistical power (1-&#x03B2;) at 80% (power=0.80) and the significance level (&#x03B1;) at 0.05. The trial was designed with 3 groups (k=3) and an allocation ratio of 1:1:1. The assumptions for means and SDs were determined with reference to the data distributions reported in 2 recent comparable studies. Wu et al [<xref ref-type="bibr" rid="ref23">23</xref>] used a Chinese question bank for hepatobiliary surgery internship teaching, where the traditional group had an average theoretical score of 77.86 (SD 4.16) and the ChatGPT o3-mini group scored 86.44 (SD 5.59). Wang et al [<xref ref-type="bibr" rid="ref24">24</xref>] conducted structured clinical tests in medical history collection training, with the baseline GPT and control groups scoring 57.39 (SD 11.14) and 54.68 (SD 10.33), respectively. Based on these data, we conservatively estimated the average score for the control group to be 75 (SD 10) and anticipated that the 2 AI groups (ChatGPT o3-mini and DeepSeek R1) would achieve an average score of 85 (SD 11). Based on a 20% dropout rate, the sample size for each group estimated by PASS (Power Analysis and Sample Size) 2021 (NCSS, LLC) was 76, resulting in a total of 228 participants, which meets the statistical requirements.</p><p>A total of 228 undergraduate students from the Medical College of Qingdao University were included in this study, ranging from the second to the fifth year. The inclusion criteria were that they were majoring in clinical medicine and had completed courses such as systemic anatomy, regional anatomy, histology, embryology, and biochemistry, with no failing records. The exclusion criteria included students who had failed courses, changed majors, used AI in the control group, or refused to participate or dropped out for personal reasons.</p></sec><sec id="s2-4"><title>Randomization</title><p>This study used a parallel-design, prospective randomized controlled trial. The allocation was indeed designed as 1:1:1. We used a sealed envelope randomization method: a total of 240 sealed envelopes were prepared, with 80 envelopes allocated to each of the 3 groups (ChatGPT, Control, and DeepSeek), reflecting the intended 1:1:1 allocation ratio. Among the 228 eligible students recruited for the study, 8 declined to participate, leaving 220 participants who were randomly assigned to 1 of the 3 groups by drawing 1 sealed envelope each. All participants were assigned by randomly drawing sealed, opaque, sequentially numbered envelopes prepared in advance with a 1:1:1 allocation ratio. The envelope opening and group assignment were conducted by an independent research coordinator who was not involved in participant recruitment or outcome assessment, ensuring no manual intervention in group assignment, no selective enrollment, and no randomization errors. The 2 models were applied to the ChatGPT o3-mini and DeepSeek R1 groups as supplementary learning tools, aiming to explore their effects and roles in helping medical undergraduates master urology knowledge. Additionally, through tests and questionnaires, we assessed students&#x2019; views and experiences regarding the application of these models in urology education.</p></sec><sec id="s2-5"><title>Learning Materials</title><p>In this study, the learning materials for participants were drawn from the National Comprehensive Cancer Network (NCCN) guidelines, the European Association of Urology (EAU) guidelines, and Chinese specialized textbooks used in higher medical education. Specifically, the study materials covered a range of urological disease types, including urinary stone disease and tumors, thereby comprehensively covering core knowledge domains in urology (<xref ref-type="table" rid="table1">Table 1</xref>). These resources, chosen for their authority and scientific rigor, underwent a rigorous selection and integration process to ensure both accuracy and current relevance. The NCCN and EAU guidelines serve as prominent international references in the field of urology and provide up-to-date diagnostic and therapeutic paradigms, whereas the Chinese medical education textbooks are more closely aligned with the participants&#x2019; domestic educational background and local clinical context. The combined use of these resources was designed to present participants with a systematic, comprehensive, and pragmatic body of knowledge, enabling a deeper understanding and mastery of key urology concepts and thereby laying a solid theoretical foundation for their future clinical practice.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Urology learning content selected for the randomized controlled trial.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Chapter and section</td><td align="left" valign="bottom">Item</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Main symptoms of urinary system</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Symptoms related to urination</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Classification of urinary incontinence</p></list-item><list-item><p>Localization analysis of hematuria</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Injury of urinary system</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Closed renal injury</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Classification</p></list-item><list-item><p>Treatment</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Urethral injury</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Classification and causes of injury</p></list-item><list-item><p>Diagnosis</p></list-item><list-item><p>Treatment</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Bladder injury</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Classification and causes of injury</p></list-item><list-item><p>Injury types</p></list-item><list-item><p>Diagnosis</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Obstruction of urinary system and urolithiasis</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Prostate hyperplasia</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Clinical manifestation</p></list-item><list-item><p>Diagnosis</p></list-item><list-item><p>Treatment</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Bladder stones</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Clinical manifestation</p></list-item><list-item><p>Diagnosis</p></list-item><list-item><p>Treatment</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Renal and ureteral stone</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Clinical manifestation</p></list-item><list-item><p>Diagnosis</p></list-item><list-item><p>Treatment</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Tuberculosis of urinary system</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Renal tuberculosis</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Etiology and pathology</p></list-item><list-item><p>Clinical manifestation</p></list-item><list-item><p>Diagnosis</p></list-item><list-item><p>Treatment</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Tumors of urinary system</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Renal tumor</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Renal carcinoma</p></list-item><list-item><p>Nephroblastoma</p></list-item><list-item><p>Renal pelvic carcinoma</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Tumors of urinary system</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Bladder cancer</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Etiology and pathology</p></list-item><list-item><p>TNM classification</p></list-item><list-item><p>Clinical manifestation</p></list-item><list-item><p>Diagnosis</p></list-item><list-item><p>Treatment</p></list-item></list></td></tr></tbody></table></table-wrap></sec><sec id="s2-6"><title>Group Design</title><p>This study used a parallel-design randomized controlled trial to assess the impact of ChatGPT o3-mini and DeepSeek R1 on learning outcomes in urology undergraduate education. Participants were stratified and randomized by grade level into 3 groups: the ChatGPT o3-mini group, the DeepSeek R1 group, and the control group. Participant allocation was performed using a grade-level stratified randomization approach combined with grouping based on prior academic performance, ensuring an equal grade distribution across the groups. Strictly unified inclusion and exclusion criteria were applied; all enrolled students had completed the core foundational medical courses with no course failures and demonstrated comparable overall academic proficiency levels. All learning materials for the participants were sourced from the materials mentioned earlier. In the ChatGPT o3-mini and DeepSeek R1 groups, students were required to use the specified AI tools (ChatGPT o3-mini or DeepSeek R1) to search for knowledge to support their learning and were not allowed to use other internet search engines. They could ask the AI questions based on the daily review content to obtain explanations of urology-related knowledge and learning suggestions in order to help deepen their understanding and memory of the review materials. To ensure optimal learning effects for participants in the experimental groups using AI tools, group coordinators collected daily usage durations through interactive conversations, guaranteeing that each participant spent at least 30 minutes per day engaging with the AI system. In this study, if students had questions or difficulty understanding AI-generated content, they could consult urologists who were invited by the researchers for clarification. Additionally, to avoid bias arising from differences in individual learning habits that may affect outcomes, we explicitly required participants in the AI groups to use identical prompt phrases. In contrast, the control group was limited to rely exclusively on traditional internet search methods, such as forums and search engines, to find relevant information for their review and was prohibited from using any AI-related software. To explain how the varying dropout rates affect the results and validity, we performed an intention-to-treat (ITT) analysis using 2 methods: worst-case imputation and multiple imputation.</p></sec><sec id="s2-7"><title>Follow-Up Strategy</title><p>The test content was derived from previous questions of both the Chinese National Medical Practitioner Examination and the United States Medical Licensing Examination and was designed to be completed within 2 hours, with a maximum score of 100 points. After the test, 2 researchers independently reviewed and scored the responses within 20 days to maintain consistency and impartiality in scoring. The primary end point of this study is total test scores of the 3 groups. The remaining measures (the scores of the 3 groups on type A1 and type A2 questions, image type questions, and text type questions) are secondary or exploratory end points. The 2 researchers who independently scored the multiple-choice tests were blinded to participants&#x2019; group allocation. Specifically, all participant identities were anonymized and recorded using numeric participant IDs only, with no information about whether the participant belonged to the DeepSeek R1, ChatGPT o3-mini, or control group. The group allocation key was kept separately by a third researcher who was not involved in scoring. Therefore, the outcome assessors were effectively blinded. Additionally, all participants in the 2 experimental groups completed a follow-up questionnaire. The design of the questionnaire was informed by relevant research literature [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>] and used a Likert scale to assess participants&#x2019; perspectives and experiences with AI-assisted learning. The questionnaires were all voluntary to fill out, and those who left any information blank were required to fill it in again to ensure an effective completion rate. The questionnaire consisted of 6 key sections. The first section gathered baseline characteristics, including participants&#x2019; gender and grade. The second section explored AI usage, focusing on its purpose, frequency, and any challenges encountered during use. The third section assessed participants&#x2019; knowledge of AI, specifically regarding its potential to alleviate the burden on both teachers and students, its ease of use, and its ability to enhance understanding and stimulate interest in learning. The fourth section addressed concerns about AI usage, examining participants&#x2019; trust in AI, their concerns, and their views on its standardized application. In the fifth section, the survey investigated participants&#x2019; opinions on the future integration of AI in undergraduate medical education, including its potential to transform teaching methods, replace educators, address resource imbalances, and integrate with traditional education models. Finally, the sixth section solicited suggestions from participants on how AI could be more effectively integrated into undergraduate medical education.</p></sec><sec id="s2-8"><title>Statistical Analysis</title><p>The age and gender of respondents and nonrespondents were compared using the 2-tailed independent samples <italic>t</italic> test and chi-square test, respectively. The survey questionnaire used frequency and percentage distributions to describe the experimental groups&#x2019; views and experiences regarding AI-assisted learning. The results of the multiple-choice test were expressed as mean (SD) to represent each group&#x2019;s performance across various question types and learning outcomes, including total scores, A1 category question scores, A2 category question scores, imaging-based question scores, and text-based question scores. Data analysis was performed using GraphPad Prism (version 9.1.0; GraphPad Software) and R (version 4.4.1; R Core Team). As the data violated the assumption of normality, comparisons among the 3 groups (Control, ChatGPT, and DeepSeek) were performed using the Kruskal-Wallis test. When the overall test was significant (<italic>P</italic>&#x003C;.05), pairwise comparisons were conducted by Dunn post hoc test with a Bonferroni correction for multiple comparisons. Effect size was quantified using &#x03B5;&#x00B2;, with 95% CIs calculated via bias-corrected and accelerated bootstrap with 1000 resamples. A <italic>P</italic> value of less than .05 was considered statistically significant.</p></sec><sec id="s2-9"><title>Primary and Sensitivity Analyses</title><p>The primary analysis was performed using ITT principles, encompassing all randomized participants (N=228). Missing test scores for the 8 participants who did not attend the examination were imputed using multiple imputation by chained equations with predictive mean matching, generating 20 imputed datasets under the missing-at-random (MAR) assumption (using the mice package in R, version 4.4.1; seed=123). To evaluate the robustness of findings, 3 sensitivity analyses were conducted: (1) a complete-case analysis, including only participants with complete outcome data (n=220); (2) a worst-case analysis, where missing values were replaced with the minimum observed score within each respective group; and (3) a best-case analysis, where missing values were replaced with the maximum observed score within each group. All analyses used linear regression with group assignment as the independent variable.</p></sec><sec id="s2-10"><title>Ethical Considerations</title><p>This study was approved by the Ethics Committee of Affiliated Hospital of Qingdao University (QYFYWZLL30898), and it was also registered with the Chinese Clinical Trials Registry (ChiCTR2600118749). The Ethics Committee of the Affiliated Hospital of Qingdao University reviewed the study protocol to ensure that participants&#x2019; personal information would not be disclosed. In accordance with the Declaration of Helsinki [<xref ref-type="bibr" rid="ref27">27</xref>], written informed consent was obtained from all participants prior to the collection of any information pertaining to them.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Participants</title><p>As of March 9, 2025, a total of 228 undergraduate students from Qingdao University Medical College were recruited to participate in the experiment. However, 8 participants withdrew before the randomization process. Additionally, 45 students from the 3 groups did not complete the multiple-choice test. Therefore, by April 5, 2025, a total of 175 students had completed the multiple-choice test (<xref ref-type="table" rid="table2">Table 2</xref>). Furthermore, 45 students from the 2 AI groups did not participate in the questionnaire survey. As a result, by April 5, 2025, a total of 105 students had completed the survey. The flowchart depicting the experimental procedure is presented in <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Baseline characteristics and test scores of medical undergraduates participating in the test.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">ChatGPT o3-mini group</td><td align="left" valign="bottom">DeepSeek R1 group</td><td align="left" valign="bottom">Control group</td></tr></thead><tbody><tr><td align="left" valign="top">Age (y), mean (SD; range)</td><td align="left" valign="top">20.68 (1.32; 19&#x2010;22)</td><td align="left" valign="top">20.70 (1.27; 19&#x2010;22)</td><td align="left" valign="top">20.64 (1.36; 19&#x2010;22)</td></tr><tr><td align="left" valign="top">Sex (male), n/N (%)</td><td align="left" valign="top">26/60 (43.33)</td><td align="left" valign="top">29/57 (50.88)</td><td align="left" valign="top">30/58 (51.72)</td></tr><tr><td align="left" valign="top">Average score, mean (SD; range)</td><td align="left" valign="top">58.67 (27.62; 12&#x2010;98)</td><td align="left" valign="top">66.14 (23.25; 10&#x2010;98)</td><td align="left" valign="top">51.83 (28.02; 14&#x2010;98)</td></tr></tbody></table></table-wrap><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Study design and flow chart. MCQs: multiple-choice questions.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89315_fig02.png"/></fig></sec><sec id="s3-2"><title>Model Characteristics and Test Questions</title><p><xref ref-type="table" rid="table3">Table 3</xref> summarizes the characteristics of DeepSeek R1 and ChatGPT o3-mini. Both models are based on the Transformer architecture [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>], but they exhibit distinct strengths shaped by their architectural designs and application foci. DeepSeek R1 focuses on cost-effectiveness, transparency, and specialized reasoning, while ChatGPT o3-mini prioritizes wide coverage, polished natural language output, and user-friendly conversations [<xref ref-type="bibr" rid="ref30">30</xref>].</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Baseline characteristics and test scores of respondents and nonrespondents on the test.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Respondents</td><td align="left" valign="bottom">Nonrespondents</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Age (y), mean (SD; range)</td><td align="left" valign="top">20.67 (1.31; 19&#x2010;22)</td><td align="left" valign="top">20.53 (1.14; 19&#x2010;22)</td><td align="left" valign="top">.46</td></tr><tr><td align="left" valign="top">Sex (male), n/N (%)</td><td align="left" valign="top">85/175 (48.57)</td><td align="left" valign="top">31/53 (58.49)</td><td align="left" valign="top">.21</td></tr></tbody></table></table-wrap><p>We selected 185 urology-related MCQs to assess the accuracy of responses from both the ChatGPT o3-mini model and the DeepSeek R1 model. These questions were categorized into various fields: anatomy (9 questions), diagnosis (84 questions), treatment (51 questions), etiology (5 questions), clinical manifestations (9 questions), complications (5 questions), and concepts (22 questions, which did not fit into the other categories). The accuracy rates for each category, as well as the overall accuracy rate, were calculated and analyzed.</p></sec><sec id="s3-3"><title>Model Response Accuracy</title><p>In the test consisting of 185 urology-related MCQs, ChatGPT o3-mini and DeepSeek R1 exhibited different accuracy performances (<xref ref-type="table" rid="table4">Table 4</xref>).</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Comparison of technical characteristics between DeepSeek R1 and ChatGPT o3-mini.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparison feature</td><td align="left" valign="bottom">DeepSeek R1</td><td align="left" valign="bottom">ChatGPT o3-mini</td></tr></thead><tbody><tr><td align="left" valign="top">Core architecture</td><td align="left" valign="top">Open-source; rule-based reinforcement learning without pre-SFT<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> [<xref ref-type="bibr" rid="ref30">30</xref>].</td><td align="left" valign="top">Proprietary; closed-source architecture [<xref ref-type="bibr" rid="ref31">31</xref>].</td></tr><tr><td align="left" valign="top">Content quality</td><td align="left" valign="top">Concise, structured; stronger completeness and currency in disease education [<xref ref-type="bibr" rid="ref32">32</xref>].</td><td align="left" valign="top">Detailed, expressive; higher clarity and accessibility for lay audiences [<xref ref-type="bibr" rid="ref33">33</xref>].</td></tr><tr><td align="left" valign="top">Privacy and deployment</td><td align="left" valign="top">Supports offline deployment [<xref ref-type="bibr" rid="ref34">34</xref>]; customizable via open-source datasets [<xref ref-type="bibr" rid="ref35">35</xref>].</td><td align="left" valign="top">Restricted by closed-source limits; challenges in health care privacy compliance [<xref ref-type="bibr" rid="ref36">36</xref>].</td></tr><tr><td align="left" valign="top">Key application strengths</td><td align="left" valign="top">Clinical decision support, specialized medical education [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref37">37</xref>].</td><td align="left" valign="top">General patient communication, broad medical education, cross-domain usability [<xref ref-type="bibr" rid="ref38">38</xref>].</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>SFT: supervised fine&#x2011;tuning.</p></fn></table-wrap-foot></table-wrap><p><xref ref-type="fig" rid="figure3">Figure 3</xref> presents the accuracy rates of the 2 models for each question type and overall, as well as the proportion of each question category (<xref ref-type="table" rid="table5">Table 5</xref>). ChatGPT o3-mini attained an overall accuracy rate of 68.11% (126/185). The accuracy rates for specific question types were as follows: etiology (4/5, 80%), concepts (13/22, 59.09%), diagnosis (60/84, 71.43%), clinical manifestations (7/9, 77.78%), complications (3/51, 60%), treatment (32/51, 62.75%), and anatomy (7/9, 77.78%). In contrast, DeepSeek R1 achieved an overall accuracy rate of 84.32% (156/185). The accuracy rates for specific question types were as follows: etiology (4/5, 80%), concepts (15/22, 68.18%), diagnosis (76/84, 90.48%), clinical manifestations (7/9, 77.78%), complications (5/5, 100%), treatment (41/51, 80.39%), and anatomy (8/9, 88.89%). DeepSeek R1 demonstrated higher accuracy across all domains compared to ChatGPT o3-mini, with particularly significant advantages in the areas of diagnosis, complications, and treatment.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>The accuracy rates of AI responses to urology multiple-choice questions and the proportion of each type of questions. Comparison between the accuracy rate of ChatGPT o3-mini and DeepSeek across (A) &#x201C;cause of disease&#x201D;, (B) &#x201C;concept&#x201D;, (C) &#x201C;diagnosis&#x201D;, (D) &#x201C;clinical manifestation&#x201D;, (E) &#x201C;complication&#x201D;, (F) &#x201C;treatment&#x201D;, (G) &#x201C;dissection&#x201D;, and (H) &#x201D;total&#x201D; question categories.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89315_fig03.png"/></fig><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Accuracy of DeepSeek R1 and ChatGPT o3-mini on 185 urology multiple-choice questions by different types.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Response category</td><td align="left" valign="bottom">Cause of disease</td><td align="left" valign="bottom">Concept</td><td align="left" valign="bottom">Diagnosis</td><td align="left" valign="bottom">Clinical manifestation</td><td align="left" valign="bottom">Complication</td><td align="left" valign="bottom">Treatment</td><td align="left" valign="bottom">Anatomy</td><td align="left" valign="bottom">Total</td></tr></thead><tbody><tr><td align="left" valign="top">Number of questions, N</td><td align="left" valign="top">5</td><td align="left" valign="top">22</td><td align="left" valign="top">84</td><td align="left" valign="top">9</td><td align="left" valign="top">5</td><td align="left" valign="top">51</td><td align="left" valign="top">9</td><td align="left" valign="top">185</td></tr><tr><td align="left" valign="top" colspan="9">ChatGPT o3-mini, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Completely correct (3 points)</td><td align="left" valign="top">4 (80)</td><td align="left" valign="top">13 (59.09)</td><td align="left" valign="top">60 (71.43)</td><td align="left" valign="top">7 (77.78)</td><td align="left" valign="top">3 (60)</td><td align="left" valign="top">32 (62.75)</td><td align="left" valign="top">7 (77.78)</td><td align="left" valign="top">126 (68.11)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>More correct than incorrect (2 points)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">5 (22.73)</td><td align="left" valign="top">5 (5.95)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">4 (7.84)</td><td align="left" valign="top">2 (22.22)</td><td align="left" valign="top">16 (8.65)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>More incorrect than correct (1 points)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">1 (4.55)</td><td align="left" valign="top">6 (7.14)</td><td align="left" valign="top">1 (11.11)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">5 (9.80)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">13 (7.03)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Completely incorrect (0 points)</td><td align="left" valign="top">1 (20)</td><td align="left" valign="top">3 (13.64)</td><td align="left" valign="top">13 (15.48)</td><td align="left" valign="top">1 (11.11)</td><td align="left" valign="top">2 (40)</td><td align="left" valign="top">10 (19.61)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">30 (16.22)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Score, mean (SD)</td><td align="left" valign="top">2.40 (1.34)</td><td align="left" valign="top">2.27 (1.08)</td><td align="left" valign="top">2.33 (1.14)</td><td align="left" valign="top">2.44 (1.13)</td><td align="left" valign="top">1.80 (1.64)</td><td align="left" valign="top">2.14 (1.23)</td><td align="left" valign="top">2.78 (0.44)</td><td align="left" valign="top">2.29 (1.15)</td></tr><tr><td align="left" valign="top" colspan="9">DeepSeek R1, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Completely correct<break/>(3 points)</td><td align="left" valign="top">4 (80)</td><td align="left" valign="top">15 (68.18)</td><td align="left" valign="top">76 (90.48)</td><td align="left" valign="top">7 (77.78)</td><td align="left" valign="top">5 (100)</td><td align="left" valign="top">41 (80.39)</td><td align="left" valign="top">8 (88.89)</td><td align="left" valign="top">156 (84.32)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>More correct than incorrect (2 points)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">1 (4.55)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">1 (0.54)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>More incorrect than correct (1 points)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Completely incorrect (0 points)</td><td align="left" valign="top">1 (20)</td><td align="left" valign="top">6 (27.27)</td><td align="left" valign="top">8 (9.52)</td><td align="left" valign="top">2 (22.22)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">10 (19.61)</td><td align="left" valign="top">1 (11.11)</td><td align="left" valign="top">28 (15.14)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Score, mean (SD)</td><td align="left" valign="top">2.40 (1.34)</td><td align="left" valign="top">2.14 (1.36)</td><td align="left" valign="top">2.71 (0.89)</td><td align="left" valign="top">2.33 (1.32)</td><td align="left" valign="top">3.00 (0)</td><td align="left" valign="top">2.41 (1.20)</td><td align="left" valign="top">2.67 (1.00)</td><td align="left" valign="top">2.54 (1.08)</td></tr></tbody></table></table-wrap></sec><sec id="s3-4"><title>Test After Self-Study</title><p>In this study, the test after self-study comprised 50 MCQs, which included 5 (10%) imaging-based questions, 26 (52%) A1-type questions, and 19 (38%) A2-type questions. A1-type questions, which were the most numerous, focused on fundamental urological knowledge to ensure students had a solid grasp of core concepts. A2-type questions assessed students&#x2019; ability to analyze clinical scenarios and solve problems. Although imaging-based questions accounted for a smaller proportion, they focused on evaluating students&#x2019; skills in interpreting imaging data, a crucial aspect of diagnosing and treating urological diseases. Overall, the test was designed to assess both theoretical knowledge and clinical practice skills, with a progressive structure that moved from basic principles to clinical application. This approach effectively aligned with the educational goal of integrating theory with practice in medical education.</p><p>The test scores of each participant across the 3 groups were recorded, and the aggregate results are shown in <xref ref-type="fig" rid="figure4">Figure 4</xref>. Kruskal-Wallis analysis demonstrated significant differences in total test scores among the 3 models (H=7.38; <italic>P</italic>=.03; &#x03B5;&#x00B2;=0.03; 95% CI 0.00-0.10; small effect). Post hoc pairwise comparisons by the Dunn test with Bonferroni correction revealed that the DeepSeek R1 group scored significantly higher than the control group (<italic>P</italic>=.02; <xref ref-type="table" rid="table6">Table 6</xref>). However, neither the comparison between ChatGPT and DeepSeek R1 nor that between ChatGPT and control reached statistical significance. For A1-type questions, a significant difference was likewise detected among the 3 groups (H=6.45; <italic>P</italic>=.04; &#x03B5;&#x00B2;=0.03; 95% CI 0.00-0.09; small effect), with DeepSeek R1 outperforming the control group (<italic>P</italic>=.04) in post hoc analysis. No other pairwise comparisons were significant. Regarding A2-type questions, the Kruskal-Wallis test yielded no significant difference across the 3 models (H=4.85; <italic>P</italic>=.09; &#x03B5;&#x00B2;=0.02; 95% CI 0.00-0.07; small effect). For imaging-based questions, significant intergroup differences emerged (H=9.19; <italic>P</italic>=.01; &#x03B5;&#x00B2;=0.04; 95% CI 0.00-0.11; small effect). Subsequent Dunn-Bonferroni testing indicated that DeepSeek R1 achieved higher scores than the control group (<italic>P</italic>=.009), whereas neither the ChatGPT vs DeepSeek R1 nor the ChatGPT vs control comparison was statistically significant. Similarly, for text-based questions, the 3 models differed significantly (H=6.49; <italic>P</italic>=.04; &#x03B5;&#x00B2;=0.03; 95% CI 0.00-0.09; small effect), with DeepSeek R1 surpassing the control group (<italic>P</italic>=.03) in pairwise comparisons; again, no significant differences were found between ChatGPT and the other 2 groups.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Test results across the 3 groups: (A) the total test scores of the 3 groups; (B) the scores of the 3 groups on type A1; (C) the scores of the 3 groups on type A2; (D) the scores of the 3 groups on image type questions; (E) the scores of the 3 groups on text type questions. *<italic>P</italic>&#x003C;.05; **<italic>P</italic>&#x003C;.01.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89315_fig04.png"/></fig><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Results of primary and sensitivity analyses.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Analysis and comparison</td><td align="left" valign="bottom">Estimate (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Intention-to-treat</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x2003;ChatGPT vs Control</named-content></td><td align="char" char="." valign="top">6.45 (&#x2013;3.74 to 16.65)</td><td align="char" char="." valign="top">.21</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek vs Control<named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content></td><td align="left" valign="top">12.65 (0.28 to 25.01)</td><td align="left" valign="top">.045</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x2003;DeepSeek vs ChatGPT</named-content></td><td align="left" valign="top">5.94 (&#x2013;6.82 to 18.70)</td><td align="left" valign="top">.35</td></tr><tr><td align="left" valign="top" colspan="3">Complete-case analysis</td></tr><tr><td align="left" valign="top">&#x2003;ChatGPT vs Control</td><td align="left" valign="top">6.84 (&#x2013;2.76 to 16.44)</td><td align="left" valign="top">.16</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek vs Control</td><td align="left" valign="top">14.31 (4.59 to 24.04)</td><td align="left" valign="top">.004</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek vs ChatGPT</td><td align="left" valign="top">7.47 (&#x2013;2.17 to 17.12&#xFF09;</td><td align="left" valign="top">.13</td></tr><tr><td align="left" valign="top" colspan="3">Worst-case scenario imputation</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ChatGPT vs Control</td><td align="left" valign="top">6.20 (&#x2013;4.06 to 16.47)</td><td align="left" valign="top">.24</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek vs Control</td><td align="left" valign="top">4.20 (&#x2013;5.74 to 14.14)</td><td align="left" valign="top">.41</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek vs ChatGPT</td><td align="left" valign="top">&#x2013;2.00 (&#x2013;11.90 to 7.90)</td><td align="left" valign="top">.69</td></tr><tr><td align="left" valign="top" colspan="3">Best-case scenario imputation</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ChatGPT vs Control</td><td align="left" valign="top">5.10 (&#x2013;4.27 to 14.47)</td><td align="left" valign="top">.29</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek vs Control</td><td align="left" valign="top">16.11 (7.04 to 25.19)</td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek vs ChatGPT<named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content></td><td align="left" valign="top">11.01 (1.97 to 20.05)</td><td align="left" valign="top">.02</td></tr></tbody></table></table-wrap></sec><sec id="s3-5"><title>Results of Primary and Sensitivity Analyses</title><p>DeepSeek R1 demonstrated significantly higher total scores than the control group (&#x03B2;=12.65, 95% CI 0.28-25.01, <italic>P</italic>=.045), whereas the difference between ChatGPT and control did not reach significance (&#x03B2;=6.45, 95% CI &#x2013;3.74 to 16.65, <italic>P</italic>=.21). The direct comparison between DeepSeek R1 and ChatGPT also showed no significant difference in the ITT analysis (&#x03B2;=5.94, 95% CI &#x2013;6.82 to 18.70, <italic>P</italic>=.35), suggesting that the observed advantage of DeepSeek R1 was primarily driven by its superiority over the control group rather than by a measurable incremental benefit over ChatGPT. Under the worst-case scenario imputation, the DeepSeek R1 advantage over control was attenuated and became nonsignificant (&#x03B2;=4.20, 95% CI &#x2013;5.74 to 14.14, <italic>P</italic>=.41), whereas under the best-case scenario, the difference was strengthened (&#x03B2;=16.11, 95% CI 7.04-25.19, <italic>P</italic>&#x003C;.001). For the DeepSeek R1 vs ChatGPT comparison, the worst-case scenario showed no significant difference (&#x03B2;=&#x2013;2.00, 95% CI &#x2013;11.90 to 7.90, <italic>P</italic>=.69), whereas the best-case scenario yielded a significant difference favoring DeepSeek R1 (&#x03B2;=11.01, 95% CI 1.97-20.05, <italic>P</italic>=.02). These findings suggest that the observed DeepSeek R1 advantage over control is maintained under standard MAR assumptions but may be partially sensitive to extreme nonrandom missingness patterns, whereas the comparison between the 2 AI models remains inconclusive regardless of missing data assumptions.</p></sec><sec id="s3-6"><title>Questionnaire Results</title><p>The questionnaire collected responses from 105 students about their perceptions of AI-assisted learning, and the main findings are presented in <xref ref-type="fig" rid="figure5">Figure 5</xref>. Regarding AI usage, most participants (n=101, 96.19%) reported using AI during their undergraduate studies. The frequency of use of GenAI by participants is shown in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. On the topic of AI&#x2019;s role in reducing the burden on both teachers and students, 44.76% (n=47) agreed and 24.76% (n=26) strongly agreed. As for AI&#x2019;s convenience, 44.76% (n=47) agreed that it allows them to seek answers anytime and anywhere, while 30.48% (n=32) strongly agreed. Furthermore, 49.52% (n=52) felt that AI helped them better understand urology-related knowledge, and 32.38% (n=34) strongly agreed. Additionally, 38.10% (n=40) agreed that AI stimulated their interest in learning urology, with 26.67% (n=28) strongly agreeing. When exploring concerns about AI use, 51.43% (n=54) of participants reported that, even if AI responses were more reliable and faster, they still preferred traditional methods for seeking answers, while 7.62% (n=8) disagreed. Regarding AI&#x2019;s prospects, all participants, except for 5 neutral responses, agreed to varying degrees that AI holds transformative potential for undergraduate medical education. Furthermore, 22.86% (n=24) agreed and 41.90% (n=44) strongly agreed that AI is capable of reshaping the prevailing model of undergraduate medical education. However, only 3.81% (n=4) believed that AI would completely supersede medical educators, while 10.48% (n=11) felt that medical educators would never be replaced by AI. When asked about participating in AI training courses offered by their institution, 73.33% (n=77) of participants expressed a willingness to attend. Additionally, 48.57% (n=51) agreed and 29.52% (n=31) strongly agreed that AI could help alleviate the imbalance in medical education resources. Finally, except for 16 neutral responses, all other participants supported, to varying degrees, the future integration of AI with traditional education models.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Undergraduate perceptions of the role of generative AI in medical education (n=105).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89315_fig05.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>Undergraduate medical education currently faces challenges such as an overwhelming curriculum, tight schedules, and students&#x2019; high cognitive load, coupled with low motivation to study [<xref ref-type="bibr" rid="ref39">39</xref>]. In this context, the introduction of GenAI has become essential. To date, several studies have explored the integration of AI into medical education. Singla et al [<xref ref-type="bibr" rid="ref40">40</xref>] used the Delphi method to develop an expert consensus-based AI curriculum framework for Canadian undergraduate medical students. Meanwhile, Wang et al [<xref ref-type="bibr" rid="ref24">24</xref>] explored the use of ChatGPT in training students&#x2019; history-taking skills. The presence of GenAI in medical education prompts a reimagining and reinterpretation of traditional roles within established pedagogy [<xref ref-type="bibr" rid="ref41">41</xref>]. On the one hand, AI can provide personalized cases instantly, based on students&#x2019; needs [<xref ref-type="bibr" rid="ref42">42</xref>], significantly boosting curiosity and engagement. It can also generate tailored study materials based on students&#x2019; learning progress and weaknesses, such as practice questions and notes, further enhancing outcomes [<xref ref-type="bibr" rid="ref43">43</xref>]. On the other hand, the instant feedback feature of AI can lead to overreliance, inhibiting critical thinking and independent analysis, and may raise concerns about academic integrity [<xref ref-type="bibr" rid="ref44">44</xref>]. Thus, AI in medical education presents a dual character [<xref ref-type="bibr" rid="ref45">45</xref>-<xref ref-type="bibr" rid="ref48">48</xref>], with its core value lying in assisting, rather than replacing, human educators. It requires the support of teacher guidance and ethical norms to truly empower the future training of medical talents [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>].</p><p>The primary objective of this study was to evaluate the effectiveness and feasibility of GenAI (ChatGPT o3-mini and DeepSeek R1) in enhancing medical undergraduates&#x2019; learning outcomes in urology. The results indicated that DeepSeek R1 had a higher accuracy rate than ChatGPT o3-mini in answering urology-related MCQs. In the test following the self-study period, the DeepSeek R1 group outperformed the control group and the ChatGPT o3-mini group, achieving higher total scores across various question types. Despite the elevated scores in the ChatGPT o3-mini cohort, statistical significance was not achieved relative to the control. Survey results showed that most students held a positive attitude toward AI-assisted learning, believing it effectively supports medical education. Currently, the urological community perceives ChatGPT and LLMs as promising tools for research [<xref ref-type="bibr" rid="ref25">25</xref>], particularly given ChatGPT&#x2019;s more empathetic responses [<xref ref-type="bibr" rid="ref16">16</xref>], but remains circumspect about associated ethical challenges and the degree of patient acceptance [<xref ref-type="bibr" rid="ref25">25</xref>].</p><p>During the study, 45 participants across all 2 groups did not take the multiple-choice test, with the majority (23 individuals) from the DeepSeek R1 group. As the control group was restricted to traditional internet search methods, which offered less novelty, participation willingness was further reduced. The students who missed the test may have had weaker academic foundations or lower motivation, potentially leading to an overestimation of the AI groups&#x2019; effectiveness. Additionally, 45 participants from the 2 AI groups did not complete the questionnaire. The high dropout rate in the questionnaire process was related to the lengthy time required for the Likert scale items and concerns about privacy in AI use. Those who did not complete the survey were more likely to hold neutral or negative views, introducing a positive bias in AI satisfaction estimates.</p><p>In terms of model accuracy, ChatGPT o3-mini showed lower accuracy on treatment and diagnosis-type questions, with rates of 62.75% (32/51) and 71.43% (60/84), respectively&#x2014;significantly lower than DeepSeek R1&#x2019;s performance. This suggests that ChatGPT o3-mini is not yet suitable as an auxiliary learning tool for treatment knowledge, but it is more appropriate for teaching fundamental medical concepts [<xref ref-type="bibr" rid="ref51">51</xref>]. While AI&#x2019;s diagnostic performance has not yet reached expert-level accuracy, recent research indicates that, if its limitations are well understood, AI may bring positive changes to medical education [<xref ref-type="bibr" rid="ref52">52</xref>]. ChatGPT o3-mini&#x2019;s overall accuracy rate of 68.11% (126/185) is comparable to the 70.60% MCQ accuracy reported for ChatGPT 4.0 in an English-language orthopedic question bank [<xref ref-type="bibr" rid="ref51">51</xref>] but still lower than DeepSeek R1&#x2019;s 84.32% (156/185). One possible explanation for this discrepancy is the language setting of this study, in which all questions were presented in Chinese. Since ChatGPT&#x2019;s training corpus predominantly consists of English content, its ability to accurately capture the logic and semantics of Chinese medical texts may be limited [<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>]. In contrast, the DeepSeek R1 model, developed by the Chinese company DeepSeek, has been specifically optimized for Chinese language processing and demonstrates stronger performance in handling medical terminology and complex reasoning in Chinese. This language difference may help explain the performance gap observed between ChatGPT o3-mini and DeepSeek R1 in this study, particularly in Chinese-language medical education settings. Future studies should systematically compare direct Chinese queries with English queries to better understand the effect of language on model performance. Such findings would help educators decide which prompt language to recommend to students when they obtain medical information from AI.</p><p>AI-assisted learning has a significant positive impact on student performance. The DeepSeek R1 group performed exceptionally well in the test following the self-study period, indicating effective support for their learning. In A1-type questions, the DeepSeek R1 group scored 37.44 points, significantly higher than the control group&#x2019;s 29 points (<italic>P</italic>=.01). Although students using AI tools demonstrated superior performance on A2-type questions, this advantage did not reach statistical significance when compared to the conventional learning group. These results suggest that AI models like DeepSeek R1 are effective in enhancing medical undergraduates&#x2019; understanding of fundamental knowledge. However, whether they can improve students&#x2019; ability to analyze urology case summaries remains unclear. In terms of text-based questions, the DeepSeek R1 group scored 60 points, significantly higher than the control group&#x2019;s 47.24 points (<italic>P</italic>=.02). Similarly, for image-based questions, the DeepSeek R1 group scored 6.14 points, surpassing the control group&#x2019;s 4.59 points (<italic>P</italic>=.01). These results suggest that AI-assisted learning not only deepens medical students&#x2019; conceptual cognition but also improves their ability to interpret imaging reports, a crucial skill in clinical practice.</p><p>The survey results indicate a strong acceptance of AI-assisted learning among students. Notably, 96.19% (101/105) of respondents reported using LLMs during their undergraduate studies, highlighting the deep integration of AI tools into the daily lives of medical students. Additionally, the majority of students believe AI can effectively reduce the workload for both teachers and students, provide convenient explanations, enhance understanding of urology concepts, and stimulate learning interest. These findings suggest that students see the potential of AI in urology education and are open to using it as a supplementary learning tool. However, some students expressed concerns about the reliability and standardization of AI-assisted learning, emphasizing the need for appropriate guidance and supervision to maximize AI&#x2019;s benefits while mitigating potential risks. Notably, 73.33% (77/105) of students are willing to participate in AI training provided by their institutions, highlighting the importance of integrating AI-related content into medical curricula to better prepare future health care professionals [<xref ref-type="bibr" rid="ref40">40</xref>]. While the rise of GenAI in medical education has sparked new perspectives on traditional teaching roles [<xref ref-type="bibr" rid="ref41">41</xref>], only 3.81% (4/105) of students believe AI will completely replace teachers, suggesting that AI is perceived as a complementary aid rather than a substitute. Additionally, 48.57% (51/105) of students believe AI could help alleviate the imbalance in educational resources, indicating that promoting AI use alongside faculty training could improve medical education, especially in underdeveloped areas. This could broaden the applicability of the study&#x2019;s conclusions and benefit medical education more widely. While AI has been the subject of intensive global investigation, its practical applications in medicine remain limited [<xref ref-type="bibr" rid="ref55">55</xref>]. The findings of this study offer empirical support to inform the future integration of AI in clinical practice and medical education. DeepSeek R1&#x2019;s diagnostic accuracy of 90.48% (76/84) in urological disease-related questions, coupled with an average response time of less than 3 seconds, indicates that the model is capable of adapting to evolving medical knowledge and performing scientific reasoning. This suggests its potential for medical education, clinical decision-making [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref25">25</xref>], and diagnosis [<xref ref-type="bibr" rid="ref56">56</xref>], as well as individualized treatment planning for urological tumors [<xref ref-type="bibr" rid="ref26">26</xref>]. However, it is critical to emphasize that accuracy on standardized multiple-choice assessments does not translate to clinical validity, diagnostic reliability, or therapeutic safety in actual patient encounters. For example, resident physicians could input symptoms, signs, and imaging descriptions to instantly obtain diagnostic suggestions and recommendations for further examinations, significantly reducing misdiagnosis rates. The integration of AI into health care shows considerable promise, particularly in elevating the quality of patient management and informing diagnostic and therapeutic choices. This will facilitate the advancement of precision medicine, which has been substantiated by numerous studies to play a pivotal role in the diagnosis and management of urological diseases [<xref ref-type="bibr" rid="ref57">57</xref>-<xref ref-type="bibr" rid="ref59">59</xref>]. However, as AI becomes increasingly integrated, ethical, legal, and accuracy-related issues must be addressed with great care [<xref ref-type="bibr" rid="ref60">60</xref>]. In real teaching scenarios, AI can assist educators in creating instructional materials, help students answer clinical questions, and promote interactive learning experiences [<xref ref-type="bibr" rid="ref60">60</xref>]. For instance, instructors could use AI models to generate reference answers before case discussions and then focus on discrepancies between student responses and model outputs to provide precise feedback. Furthermore, survey results indicate that 73.33% (77/105) of students are willing to engage in AI training organized by their institutions, underscoring the potential to integrate AI literacy and critical thinking into core medical curricula [<xref ref-type="bibr" rid="ref61">61</xref>]. Existing evidence suggests that the integration of GenAI into medical learning environments carries an acceptable safety profile, thereby supporting its further exploration. We should embrace the changes it brings to medical education with an open mindset and proactively integrate it into lifelong medical learning processes [<xref ref-type="bibr" rid="ref62">62</xref>]. If implemented thoughtfully, GenAI has the potential to significantly support clinicians in improving medical quality, enhancing medical education, reducing workloads, and providing real-time professional knowledge support [<xref ref-type="bibr" rid="ref63">63</xref>].</p><p>It is important to emphasize that the application of the results from this study must address the ethical considerations associated with the use of AI in medical education. From a data privacy standpoint, the training and validation of AI tools require access to large volumes of clinical and imaging data. This raises concerns related to data collection, transmission, storage, and the need to obtain informed consent [<xref ref-type="bibr" rid="ref64">64</xref>]. If these data are not properly protected, the risk of breaches increases, underscoring the necessity of implementing stringent data protection measures and establishing clear guidelines for their usage [<xref ref-type="bibr" rid="ref65">65</xref>]. Furthermore, attention should be paid to the authority of information and the risk of misleading. Although DeepSeek R1 demonstrated a relatively high accuracy rate in this study, AI-generated content still has the phenomenon of hallucination, meaning that the generated information may seem reasonable but is actually incorrect or misleading [<xref ref-type="bibr" rid="ref66">66</xref>,<xref ref-type="bibr" rid="ref67">67</xref>]. In medical education, such misinformation can mislead students, resulting in the formation of incorrect knowledge perceptions that may pose risks to future clinical decision-making. Educators must therefore assume a leading role in establishing comprehensive guidelines to govern the incorporation of AI within educational frameworks. These guidelines should foster critical thinking among students rather than passively relying on AI-generated answers [<xref ref-type="bibr" rid="ref68">68</xref>].</p><p>This study makes distinct contributions to the existing literature. First, it simultaneously evaluates the effectiveness of 2 AI models (ChatGPT o3-mini and DeepSeek R1) in medical education, offering a more comprehensive comparison of different AI tools within the educational domain. Second, it not only examines the accuracy of AI in answering medical questions but also explores the effects of AI on students&#x2019; learning outcomes and attitudes through testing and questionnaire surveys.</p><p>However, this study also has several limitations. First, the sample size is relatively small and limited to students from a single medical college, which may limit the broader applicability of the findings. Subsequent studies are recommended to encompass larger cohorts through multicenter randomized controlled trials, thereby improving the generalizability of the findings. Second, the study duration was relatively short, and the enduring impacts of AI-assisted learning on students&#x2019; knowledge retention and clinical skill development remain unexplored. Future studies could extend the research period to better assess the enduring effects of AI in medical education. Third, this study focused solely on the field of urology within medical education. Future research could expand the scope to include other medical specialties, thereby providing more comprehensive evidence to support the broader application of AI in medical education. Fourth, the questionnaire was administered only to respondents; nonrespondents may hold less favorable views of AI. AI groups were restricted to the assigned tools without access to other search engines. Thus, observed differences may reflect variations in task structure, information format, cognitive demand, and novelty effects. For example, the conversational single-interface interaction may have been more engaging than traditional search, while the inability to cross-check AI outputs may have increased susceptibility to incorrect information. We advocate for the integration of faculty-mediated review protocols, iterative feedback loops, and mandatory AI literacy modules into urology curricula prior to broader deployment. Furthermore, we acknowledge that the DeepSeek R1 group experienced a higher attrition rate. This differential dropout raises the possibility of survivorship bias, whereby more motivated or technologically proficient students may have been more likely to complete the DeepSeek R1 intervention. Our ITT analysis under the MAR assumption supported the primary findings; however, the worst-case scenario imputation attenuated the DeepSeek R1 advantage to nonsignificance, suggesting that the result may be partially sensitive to the performance level of dropouts. Future studies should use strategies to minimize differential attrition, such as enhanced training, technical support, and reminder systems for novel AI models.</p></sec><sec id="s4-2"><title>Conclusion</title><p>ChatGPT o3-mini and DeepSeek R1 demonstrated high accuracy in answering questions related to urology. DeepSeek-assisted self-study was linked to better posttest performance than conventional online learning; by contrast, the higher scores observed with ChatGPT were only numerical and lacked statistical significance. Most students demonstrated strong receptivity to AI-assisted learning and expressed willingness to participate in AI training courses organized by their institution, suggesting that it is highly necessary to integrate AI into medical courses in the future. The findings of this study provide evidence-based references for the integration of GenAI in medical education and offer guidance for educators in formulating teaching strategies and for educational institutions in formulating policies.</p></sec></sec></body><back><ack><p>The authors extend their appreciation to all participating students from Qingdao University Medical College for their involvement in the study and follow-up survey. ChatGPT and DeepSeek, the large language models evaluated in this trial, were used only as study experimental interventions. Neither of these models nor any other generative AI tools were used to draft, compose, or edit any part of this manuscript. All authors take full responsibility for the accuracy, originality, and integrity of the manuscript, including all references and citations.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec><sec><title>Data Availability</title><p>The data can be obtained from the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>WY and TX were involved in the study conception and design. WY and JW supervised data collection. JW and WZ performed material preparation, data collection, and analysis. WY wrote the first draft of the manuscript, and all authors reviewed and edited subsequent versions. GC and HN made crucial contributions to manuscript revision, including supplementary analyses, responses to reviewers, and manuscript polishing. WY and TX contributed equally. GC and HN were the corresponding authors. GC is the co-corresponding author (email: chuguangdi1997@126.com). All authors read and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CONSORT-EHEALTH</term><def><p>Consolidated Standards of Reporting Trials of Electronic and Mobile Health Applications and Online Telehealth.</p></def></def-item><def-item><term id="abb2">EAU</term><def><p>European Association of Urology</p></def></def-item><def-item><term id="abb3">GenAI</term><def><p>generative AI</p></def></def-item><def-item><term id="abb4">ITT</term><def><p>intention-to-treat</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">MAR</term><def><p>missing-at-random</p></def></def-item><def-item><term id="abb7">MCQ</term><def><p>multiple-choice question</p></def></def-item><def-item><term id="abb8">NCCN </term><def><p>National Comprehensive Cancer Network</p></def></def-item><def-item><term id="abb9">PASS</term><def><p>Power Analysis and Sample Size</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deeb</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gangadhar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rabindranath</surname><given-names>M</given-names> </name><etal/></person-group><article-title>The emerging role of generative artificial intelligence in transplant medicine</article-title><source>Am J Transplant</source><year>2024</year><month>10</month><volume>24</volume><issue>10</issue><fpage>1724</fpage><lpage>1730</lpage><pub-id pub-id-type="doi">10.1016/j.ajt.2024.06.009</pub-id><pub-id pub-id-type="medline">38901561</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name></person-group><article-title>Reflection on whether Chat GPT should be banned by academia from the perspective of education and teaching</article-title><source>Front Psychol</source><year>2023</year><volume>14</volume><fpage>1181712</fpage><pub-id pub-id-type="doi">10.3389/fpsyg.2023.1181712</pub-id><pub-id pub-id-type="medline">37325766</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Atchley</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Vukic</surname><given-names>B</given-names> </name><name name-style="western"><surname>Vukic</surname><given-names>M</given-names> </name><name name-style="western"><surname>Walters</surname><given-names>BC</given-names> </name></person-group><article-title>Review of cerebrospinal fluid physiology and dynamics: a call for medical education reform</article-title><source>Neurosurgery</source><year>2022</year><month>07</month><day>1</day><volume>91</volume><issue>1</issue><fpage>1</fpage><lpage>7</lpage><pub-id pub-id-type="doi">10.1227/neu.0000000000002000</pub-id><pub-id pub-id-type="medline">35522666</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Malau-Aduli</surname><given-names>BS</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>AY</given-names> </name><name name-style="western"><surname>Cooling</surname><given-names>N</given-names> </name><name name-style="western"><surname>Catchpole</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jose</surname><given-names>M</given-names> </name><name name-style="western"><surname>Turner</surname><given-names>R</given-names> </name></person-group><article-title>Retention of knowledge and perceived relevance of basic sciences in an integrated case-based learning (CBL) curriculum</article-title><source>BMC Med Educ</source><year>2013</year><month>10</month><day>8</day><volume>13</volume><fpage>139</fpage><pub-id pub-id-type="doi">10.1186/1472-6920-13-139</pub-id><pub-id pub-id-type="medline">24099045</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Long</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name></person-group><article-title>Virtual simulation in undergraduate medical education: a scoping review of recent practice</article-title><source>Front Med (Lausanne)</source><year>2022</year><volume>9</volume><fpage>855403</fpage><pub-id pub-id-type="doi">10.3389/fmed.2022.855403</pub-id><pub-id pub-id-type="medline">35433717</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wong</surname><given-names>G</given-names> </name><name name-style="western"><surname>Greenhalgh</surname><given-names>T</given-names> </name><name name-style="western"><surname>Pawson</surname><given-names>R</given-names> </name></person-group><article-title>Internet-based medical education: a realist review of what works, for whom and in what circumstances</article-title><source>BMC Med Educ</source><year>2010</year><month>02</month><day>2</day><volume>10</volume><fpage>12</fpage><pub-id pub-id-type="doi">10.1186/1472-6920-10-12</pub-id><pub-id pub-id-type="medline">20122253</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Venkatesan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mohan</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ryan</surname><given-names>JR</given-names> </name><etal/></person-group><article-title>Virtual and augmented reality for biomedical applications</article-title><source>Cell Rep Med</source><year>2021</year><month>07</month><day>20</day><volume>2</volume><issue>7</issue><fpage>100348</fpage><pub-id pub-id-type="doi">10.1016/j.xcrm.2021.100348</pub-id><pub-id pub-id-type="medline">34337564</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gan</surname><given-names>W</given-names> </name><name name-style="western"><surname>Mok</surname><given-names>TN</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Researching the application of virtual reality in medical education: one-year follow-up of a randomized trial</article-title><source>BMC Med Educ</source><year>2023</year><month>01</month><day>3</day><volume>23</volume><issue>1</issue><fpage>3</fpage><pub-id pub-id-type="doi">10.1186/s12909-022-03992-6</pub-id><pub-id pub-id-type="medline">36597093</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Temsah</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Alhuzaimi</surname><given-names>AN</given-names> </name><name name-style="western"><surname>Almansour</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Art or artifact: evaluating the accuracy, appeal, and educational value of AI-generated imagery in DALL&#x00B7;E 3 for illustrating congenital heart diseases</article-title><source>J Med Syst</source><year>2024</year><month>05</month><day>23</day><volume>48</volume><issue>1</issue><fpage>54</fpage><pub-id pub-id-type="doi">10.1007/s10916-024-02072-0</pub-id><pub-id pub-id-type="medline">38780839</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abualadas</surname><given-names>HM</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>L</given-names> </name></person-group><article-title>Achievement of learning outcomes in non-traditional (online) versus traditional (face-to-face) anatomy teaching in medical schools: a mixed method systematic review</article-title><source>Clin Anat</source><year>2023</year><month>01</month><volume>36</volume><issue>1</issue><fpage>50</fpage><lpage>76</lpage><pub-id pub-id-type="doi">10.1002/ca.23942</pub-id><pub-id pub-id-type="medline">35969356</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kung</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Cheatham</surname><given-names>M</given-names> </name><name name-style="western"><surname>Medenilla</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models</article-title><source>PLOS Digit Health</source><year>2023</year><month>02</month><volume>2</volume><issue>2</issue><fpage>e0000198</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000198</pub-id><pub-id pub-id-type="medline">36812645</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>H</given-names> </name></person-group><article-title>The rise of ChatGPT: exploring its potential in medical education</article-title><source>Anat Sci Educ</source><year>2024</year><volume>17</volume><issue>5</issue><fpage>926</fpage><lpage>931</lpage><pub-id pub-id-type="doi">10.1002/ase.2270</pub-id><pub-id pub-id-type="medline">36916887</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><etal/></person-group><article-title>The application of large language models in medicine: a scoping review</article-title><source>iScience</source><year>2024</year><month>05</month><day>17</day><volume>27</volume><issue>5</issue><fpage>109713</fpage><pub-id pub-id-type="doi">10.1016/j.isci.2024.109713</pub-id><pub-id pub-id-type="medline">38746668</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ahn</surname><given-names>C</given-names> </name></person-group><article-title>Exploring ChatGPT for information of cardiopulmonary resuscitation</article-title><source>Resuscitation</source><year>2023</year><month>04</month><volume>185</volume><fpage>109729</fpage><pub-id pub-id-type="doi">10.1016/j.resuscitation.2023.109729</pub-id><pub-id pub-id-type="medline">36773836</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gupta</surname><given-names>R</given-names> </name><name name-style="western"><surname>Herzog</surname><given-names>I</given-names> </name><name name-style="western"><surname>Park</surname><given-names>JB</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT on the plastic surgery inservice training examination</article-title><source>Aesthet Surg J</source><year>2023</year><month>11</month><day>16</day><volume>43</volume><issue>12</issue><fpage>NP1078</fpage><lpage>NP1082</lpage><pub-id pub-id-type="doi">10.1093/asj/sjad128</pub-id><pub-id pub-id-type="medline">37128784</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ayers</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Poliak</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dredze</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Comparing physician and artificial intelligence chatbot responses to patient questions posted to a public social media forum</article-title><source>JAMA Intern Med</source><year>2023</year><month>06</month><day>1</day><volume>183</volume><issue>6</issue><fpage>589</fpage><lpage>596</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2023.1838</pub-id><pub-id pub-id-type="medline">37115527</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arif</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Munaf</surname><given-names>U</given-names> </name><name name-style="western"><surname>Ul-Haque</surname><given-names>I</given-names> </name></person-group><article-title>The future of medical education and research: is ChatGPT a blessing or blight in disguise?</article-title><source>Med Educ Online</source><year>2023</year><month>12</month><volume>28</volume><issue>1</issue><fpage>2181052</fpage><pub-id pub-id-type="doi">10.1080/10872981.2023.2181052</pub-id><pub-id pub-id-type="medline">36809073</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>The Lancet Digital Health</collab></person-group><article-title>ChatGPT: friend or foe?</article-title><source>Lancet Digit Health</source><year>2023</year><month>03</month><volume>5</volume><issue>3</issue><fpage>e102</fpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00023-7</pub-id><pub-id pub-id-type="medline">36754723</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kr&#x00FC;gel</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ostermaier</surname><given-names>A</given-names> </name><name name-style="western"><surname>Uhl</surname><given-names>M</given-names> </name></person-group><article-title>ChatGPT&#x2019;s inconsistent moral advice influences users&#x2019; judgment</article-title><source>Sci Rep</source><year>2023</year><month>04</month><day>6</day><volume>13</volume><issue>1</issue><fpage>4569</fpage><pub-id pub-id-type="doi">10.1038/s41598-023-31341-0</pub-id><pub-id pub-id-type="medline">37024502</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Agathokleous</surname><given-names>E</given-names> </name><name name-style="western"><surname>Saitanis</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>Z</given-names> </name></person-group><article-title>Use of ChatGPT: what does it mean for biology and environmental science?</article-title><source>Sci Total Environ</source><year>2023</year><month>08</month><day>25</day><volume>888</volume><fpage>164154</fpage><pub-id pub-id-type="doi">10.1016/j.scitotenv.2023.164154</pub-id><pub-id pub-id-type="medline">37201835</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stokel-Walker</surname><given-names>C</given-names> </name><name name-style="western"><surname>Van Noorden</surname><given-names>R</given-names> </name></person-group><article-title>What ChatGPT and generative AI mean for science</article-title><source>Nature</source><year>2023</year><month>02</month><volume>614</volume><issue>7947</issue><fpage>214</fpage><lpage>216</lpage><pub-id pub-id-type="doi">10.1038/d41586-023-00340-6</pub-id><pub-id pub-id-type="medline">36747115</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name><etal/></person-group><article-title>The performance of large language model-powered chatbots compared to oncology physicians on colorectal cancer queries</article-title><source>Int J Surg</source><year>2024</year><month>10</month><day>1</day><volume>110</volume><issue>10</issue><fpage>6509</fpage><lpage>6517</lpage><pub-id pub-id-type="doi">10.1097/JS9.0000000000001850</pub-id><pub-id pub-id-type="medline">38935100</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Han</surname><given-names>M</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>N</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>C</given-names> </name></person-group><article-title>Application of ChatGPT-based blended medical teaching in clinical education of hepatobiliary surgery</article-title><source>Med Teach</source><year>2025</year><month>03</month><volume>47</volume><issue>3</issue><fpage>445</fpage><lpage>449</lpage><pub-id pub-id-type="doi">10.1080/0142159X.2024.2339412</pub-id><pub-id pub-id-type="medline">38614458</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Fan</surname><given-names>TT</given-names> </name><name name-style="western"><surname>Li</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>XC</given-names> </name></person-group><article-title>Feasibility study of using GPT for history-taking training in medical education: a randomized clinical trial</article-title><source>BMC Med Educ</source><year>2025</year><month>07</month><day>10</day><volume>25</volume><issue>1</issue><fpage>1030</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-07614-9</pub-id><pub-id pub-id-type="medline">40640776</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eppler</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ganjavi</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ramacciotti</surname><given-names>LS</given-names> </name><etal/></person-group><article-title>Awareness and use of ChatGPT and large language models: a prospective cross-sectional global survey in urology</article-title><source>Eur Urol</source><year>2024</year><month>02</month><volume>85</volume><issue>2</issue><fpage>146</fpage><lpage>153</lpage><pub-id pub-id-type="doi">10.1016/j.eururo.2023.10.014</pub-id><pub-id pub-id-type="medline">37926642</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rodler</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kopliku</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ulrich</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Patients&#x2019; trust in artificial intelligence-based decision-making for localized prostate cancer: results from a prospective trial</article-title><source>Eur Urol Focus</source><year>2024</year><month>07</month><volume>10</volume><issue>4</issue><fpage>654</fpage><lpage>661</lpage><pub-id pub-id-type="doi">10.1016/j.euf.2023.10.020</pub-id><pub-id pub-id-type="medline">37923632</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="web"><article-title>WMA Declaration of Helsinki &#x2013; ethical principles for medical research involving human participants</article-title><source>World Medical Association</source><access-date>2026-09-08</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.wma.net/policies-post/wma-declaration-of-helsinki">https://www.wma.net/policies-post/wma-declaration-of-helsinki</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naghdi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>P</given-names> </name><name name-style="western"><surname>Essers</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Artificial intelligence-simplified information to advance reproductive genetic literacy and health equity</article-title><source>Hum Reprod</source><year>2025</year><month>09</month><day>1</day><volume>40</volume><issue>9</issue><fpage>1681</fpage><lpage>1688</lpage><pub-id pub-id-type="doi">10.1093/humrep/deaf135</pub-id><pub-id pub-id-type="medline">40692125</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>E</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>G</given-names> </name><etal/></person-group><article-title>A highly scalable deep learning language model for common risks prediction among psychiatric inpatients</article-title><source>BMC Med</source><year>2025</year><month>05</month><day>28</day><volume>23</volume><issue>1</issue><fpage>308</fpage><pub-id pub-id-type="doi">10.1186/s12916-025-04150-7</pub-id><pub-id pub-id-type="medline">40437564</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>I</given-names> </name><name name-style="western"><surname>Tangsrivimol</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Darzi</surname><given-names>E</given-names> </name><etal/></person-group><article-title>DeepSeek vs. ChatGPT: prospects and challenges</article-title><source>Front Artif Intell</source><year>2025</year><volume>8</volume><fpage>1576992</fpage><pub-id pub-id-type="doi">10.3389/frai.2025.1576992</pub-id><pub-id pub-id-type="medline">40612384</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rouphail</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hillis</surname><given-names>E</given-names> </name><etal/></person-group><article-title>EchoLLM: extracting echocardiogram entities with light-weight, open-source large language models</article-title><source>JAMIA Open</source><year>2025</year><month>08</month><volume>8</volume><issue>4</issue><fpage>ooaf092</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooaf092</pub-id><pub-id pub-id-type="medline">40809469</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Assessing the role of large language models between ChatGPT and DeepSeek in asthma education for bilingual individuals: comparative study</article-title><source>JMIR Med Inform</source><year>2025</year><month>08</month><day>13</day><volume>13</volume><fpage>e65365</fpage><pub-id pub-id-type="doi">10.2196/65365</pub-id><pub-id pub-id-type="medline">40802989</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ehling-Schulz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Filter</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zinsstag</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Risk negotiation: a framework for One Health risk analysis</article-title><source>Bull World Health Organ</source><year>2024</year><month>06</month><day>1</day><volume>102</volume><issue>6</issue><fpage>453</fpage><lpage>456</lpage><pub-id pub-id-type="doi">10.2471/BLT.23.290672</pub-id><pub-id pub-id-type="medline">38812798</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>W</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>F</given-names> </name></person-group><article-title>DeepSeek-assisted LI-RADS classification: AI-driven precision in hepatocellular carcinoma diagnosis</article-title><source>Int J Surg</source><year>2025</year><volume>111</volume><issue>9</issue><fpage>5970</fpage><lpage>5979</lpage><pub-id pub-id-type="doi">10.1097/JS9.0000000000002763</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>ElSayed</surname><given-names>A</given-names> </name><name name-style="western"><surname>Updegrove</surname><given-names>GF</given-names> </name></person-group><article-title>Limitations of broadly trained LLMs in interpreting orthopedic Walch glenoid classifications</article-title><source>Front Artif Intell</source><year>2025</year><volume>8</volume><fpage>1644093</fpage><pub-id pub-id-type="doi">10.3389/frai.2025.1644093</pub-id><pub-id pub-id-type="medline">40951327</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alsadhan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Al-Anezi</surname><given-names>F</given-names> </name><name name-style="western"><surname>Almohanna</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The opportunities and challenges of adopting ChatGPT in medical research</article-title><source>Front Med (Lausanne)</source><year>2023</year><volume>10</volume><fpage>1259640</fpage><pub-id pub-id-type="doi">10.3389/fmed.2023.1259640</pub-id><pub-id pub-id-type="medline">38188345</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Temizsoy Korkmaz</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ok</surname><given-names>F</given-names> </name><name name-style="western"><surname>Karip</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kele&#x015F;</surname><given-names>P</given-names> </name></person-group><article-title>A structured evaluation of LLM-generated step-by-step instructions in cadaveric brachial plexus dissection</article-title><source>BMC Med Educ</source><year>2025</year><month>07</month><day>1</day><volume>25</volume><issue>1</issue><fpage>903</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-07493-0</pub-id><pub-id pub-id-type="medline">40598351</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Sanders</surname><given-names>HM</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>ChatGPT: promise and challenges for deployment in low- and middle-income countries</article-title><source>Lancet Reg Health West Pac</source><year>2023</year><month>12</month><volume>41</volume><fpage>100905</fpage><pub-id pub-id-type="doi">10.1016/j.lanwpc.2023.100905</pub-id><pub-id pub-id-type="medline">37731897</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name></person-group><article-title>The effect of career calling on medicine students&#x2019; learning engagement: chain mediation roles of career decision self-efficacy and career adaptability</article-title><source>Front Med (Lausanne)</source><year>2024</year><volume>11</volume><fpage>1418879</fpage><pub-id pub-id-type="doi">10.3389/fmed.2024.1418879</pub-id><pub-id pub-id-type="medline">39664318</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singla</surname><given-names>R</given-names> </name><name name-style="western"><surname>Pupic</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ghaffarizadeh</surname><given-names>SA</given-names> </name><etal/></person-group><article-title>Developing a Canadian artificial intelligence medical curriculum using a Delphi study</article-title><source>NPJ Digit Med</source><year>2024</year><month>11</month><day>18</day><volume>7</volume><issue>1</issue><fpage>323</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01307-1</pub-id><pub-id pub-id-type="medline">39557985</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Balasooriya</surname><given-names>C</given-names> </name><name name-style="western"><surname>Semmler</surname><given-names>C</given-names> </name><name name-style="western"><surname>Rhee</surname><given-names>J</given-names> </name></person-group><article-title>Generative artificial intelligence: the &#x201C;more knowledgeable other&#x201D; in a social constructivist framework of medical education</article-title><source>NPJ Digit Med</source><year>2025</year><month>07</month><day>11</day><volume>8</volume><issue>1</issue><fpage>430</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01823-8</pub-id><pub-id pub-id-type="medline">40646156</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>J</given-names> </name></person-group><article-title>Prompts, privacy, and personalized learning: integrating AI into nursing education-a qualitative study</article-title><source>BMC Nurs</source><year>2025</year><month>04</month><day>29</day><volume>24</volume><issue>1</issue><fpage>470</fpage><pub-id pub-id-type="doi">10.1186/s12912-025-03115-8</pub-id><pub-id pub-id-type="medline">40301862</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tzeng</surname><given-names>SY</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>KY</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>CY</given-names> </name></person-group><article-title>Predicting college students&#x2019; adoption of technology for self-directed learning: a model based on the theory of planned behavior with self-evaluation as an intermediate variable</article-title><source>Front Psychol</source><year>2022</year><volume>13</volume><fpage>865803</fpage><pub-id pub-id-type="doi">10.3389/fpsyg.2022.865803</pub-id><pub-id pub-id-type="medline">35615179</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naseer</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Saeed</surname><given-names>S</given-names> </name><name name-style="western"><surname>Afzal</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>S</given-names> </name><name name-style="western"><surname>Malik</surname><given-names>MGR</given-names> </name></person-group><article-title>Navigating the integration of artificial intelligence in the medical education curriculum: a mixed-methods study exploring the perspectives of medical students and faculty in Pakistan</article-title><source>BMC Med Educ</source><year>2025</year><month>02</month><day>20</day><volume>25</volume><issue>1</issue><fpage>273</fpage><pub-id pub-id-type="doi">10.1186/s12909-024-06552-2</pub-id><pub-id pub-id-type="medline">39979912</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Heacock</surname><given-names>L</given-names> </name><name name-style="western"><surname>Elias</surname><given-names>J</given-names> </name><etal/></person-group><article-title>ChatGPT and other large language models are double-edged swords</article-title><source>Radiology</source><year>2023</year><month>04</month><volume>307</volume><issue>2</issue><fpage>e230163</fpage><pub-id pub-id-type="doi">10.1148/radiol.230163</pub-id><pub-id pub-id-type="medline">36700838</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gordijn</surname><given-names>B</given-names> </name><name name-style="western"><surname>Have</surname><given-names>HT</given-names> </name></person-group><article-title>ChatGPT: evolution or revolution?</article-title><source>Med Health Care Philos</source><year>2023</year><month>03</month><volume>26</volume><issue>1</issue><fpage>1</fpage><lpage>2</lpage><pub-id pub-id-type="doi">10.1007/s11019-023-10136-0</pub-id><pub-id pub-id-type="medline">36656495</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><article-title>Tools such as ChatGPT threaten transparent science; here are our ground rules for their use</article-title><source>Nature</source><year>2023</year><month>01</month><day>26</day><volume>613</volume><issue>7945</issue><fpage>612</fpage><lpage>612</lpage><pub-id pub-id-type="doi">10.1038/d41586-023-00191-1</pub-id><pub-id pub-id-type="medline">36694020</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Bai</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Exploring the application capability of ChatGPT as an instructor in skills education for dental medical students: randomized controlled trial</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>27</day><volume>27</volume><fpage>e68538</fpage><pub-id pub-id-type="doi">10.2196/68538</pub-id><pub-id pub-id-type="medline">40424023</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ahmad</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Han</surname><given-names>H</given-names> </name><name name-style="western"><surname>Alam</surname><given-names>MM</given-names> </name><etal/></person-group><article-title>Retracted article: impact of artificial intelligence on human loss in decision making, laziness and safety in education</article-title><source>Humanit Soc Sci Commun</source><year>2023</year><volume>10</volume><issue>1</issue><fpage>311</fpage><pub-id pub-id-type="doi">10.1057/s41599-023-01787-8</pub-id><pub-id pub-id-type="medline">37325188</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bai</surname><given-names>R</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>T</given-names> </name></person-group><article-title>Teaching innovation in a pharmacy course: integration of &#x201C;Questioning-Training of Comprehensive Knowledge Application&#x201D; and a &#x201C;Teacher-AI-Student Interaction Model&#x201D;</article-title><source>BMC Med Educ</source><year>2025</year><month>07</month><day>1</day><volume>25</volume><issue>1</issue><fpage>964</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-07549-1</pub-id><pub-id pub-id-type="medline">40596980</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gan</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ouyang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Integrating ChatGPT in orthopedic education for medical undergraduates: randomized controlled trial</article-title><source>J Med Internet Res</source><year>2024</year><month>08</month><day>20</day><volume>26</volume><fpage>e57037</fpage><pub-id pub-id-type="doi">10.2196/57037</pub-id><pub-id pub-id-type="medline">39163598</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Takita</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kabata</surname><given-names>D</given-names> </name><name name-style="western"><surname>Walston</surname><given-names>SL</given-names> </name><etal/></person-group><article-title>A systematic review and meta-analysis of diagnostic performance comparison between generative AI and physicians</article-title><source>NPJ Digit Med</source><year>2025</year><month>03</month><day>22</day><volume>8</volume><issue>1</issue><fpage>175</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01543-z</pub-id><pub-id pub-id-type="medline">40121370</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>W</given-names> </name><etal/></person-group><article-title>How does ChatGPT-4 preform on non-English national medical licensing examination? An evaluation in Chinese language</article-title><source>PLOS Digit Health</source><year>2023</year><month>12</month><volume>2</volume><issue>12</issue><fpage>e0000397</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000397</pub-id><pub-id pub-id-type="medline">38039286</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zong</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>E</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>B</given-names> </name></person-group><article-title>Performance of ChatGPT on Chinese national medical licensing examinations: a five-year examination evaluation study for physicians, pharmacists and nurses</article-title><source>BMC Med Educ</source><year>2024</year><month>02</month><day>14</day><volume>24</volume><issue>1</issue><fpage>143</fpage><pub-id pub-id-type="doi">10.1186/s12909-024-05125-7</pub-id><pub-id pub-id-type="medline">38355517</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Nair</surname><given-names>SV</given-names> </name><name name-style="western"><surname>Ke</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Disparities in clinical studies of AI enabled applications from a global perspective</article-title><source>NPJ Digit Med</source><year>2024</year><month>08</month><day>10</day><volume>7</volume><issue>1</issue><fpage>209</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01212-7</pub-id><pub-id pub-id-type="medline">39127820</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fiorentino</surname><given-names>V</given-names> </name><name name-style="western"><surname>Pizzimenti</surname><given-names>C</given-names> </name><name name-style="western"><surname>Franchina</surname><given-names>M</given-names> </name><etal/></person-group><article-title>The minefield of indeterminate thyroid nodules: could artificial intelligence be a suitable diagnostic tool?</article-title><source>Diagn Histopathol</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>396</fpage><lpage>401</lpage><pub-id pub-id-type="doi">10.1016/j.mpdhp.2023.06.013</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fiorentino</surname><given-names>V</given-names> </name><name name-style="western"><surname>Pepe</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zuccal&#x00E0;</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Gleason score down and upgrading at radical prostatectomy in targeted vs. systematic prostate biopsy: findings from an institutional cohort</article-title><source>Pathol Res Pract</source><year>2025</year><month>07</month><volume>271</volume><fpage>156040</fpage><pub-id pub-id-type="doi">10.1016/j.prp.2025.156040</pub-id><pub-id pub-id-type="medline">40446479</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pepe</surname><given-names>P</given-names> </name><name name-style="western"><surname>Pepe</surname><given-names>L</given-names> </name><name name-style="western"><surname>Fiorentino</surname><given-names>V</given-names> </name><name name-style="western"><surname>Curduman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Pennisi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Fraggetta</surname><given-names>F</given-names> </name></person-group><article-title>PSMA PET/CT accuracy in diagnosing prostate cancer nodes metastases</article-title><source>In Vivo</source><year>2024</year><volume>38</volume><issue>6</issue><fpage>2880</fpage><lpage>2885</lpage><pub-id pub-id-type="doi">10.21873/invivo.13769</pub-id><pub-id pub-id-type="medline">39477430</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pepe</surname><given-names>P</given-names> </name><name name-style="western"><surname>Pepe</surname><given-names>L</given-names> </name><name name-style="western"><surname>Fiorentino</surname><given-names>V</given-names> </name><name name-style="western"><surname>Curduman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Fraggetta</surname><given-names>F</given-names> </name></person-group><article-title>Multiparametric MRI targeted prostate biopsy: when omit systematic biopsy?</article-title><source>Arch Ital Urol Androl</source><year>2024</year><month>11</month><day>11</day><volume>96</volume><issue>4</issue><fpage>12992</fpage><pub-id pub-id-type="doi">10.4081/aiua.2024.12992</pub-id><pub-id pub-id-type="medline">39692414</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Iqbal</surname><given-names>U</given-names> </name><name name-style="western"><surname>Tanweer</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rahmanti</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Greenfield</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>LTJ</given-names> </name><name name-style="western"><surname>Li</surname><given-names>YCJ</given-names> </name></person-group><article-title>Impact of large language model (ChatGPT) in healthcare: an umbrella review and evidence synthesis</article-title><source>J Biomed Sci</source><year>2025</year><month>05</month><day>7</day><volume>32</volume><issue>1</issue><fpage>45</fpage><pub-id pub-id-type="doi">10.1186/s12929-025-01131-z</pub-id><pub-id pub-id-type="medline">40335969</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rainey</surname><given-names>C</given-names> </name><name name-style="western"><surname>O&#x2019;Regan</surname><given-names>T</given-names> </name><name name-style="western"><surname>Matthew</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Beauty Is in the AI of the beholder: are we ready for the clinical integration of artificial intelligence in radiography? An exploratory analysis of perceived AI knowledge, skills, confidence, and education perspectives of UK radiographers</article-title><source>Front Digit Health</source><year>2021</year><volume>3</volume><fpage>739327</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2021.739327</pub-id><pub-id pub-id-type="medline">34859245</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xin</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>D</given-names> </name></person-group><article-title>ChatGPT in medicine: prospects and challenges: a review article</article-title><source>Int J Surg</source><year>2024</year><month>06</month><day>1</day><volume>110</volume><issue>6</issue><fpage>3701</fpage><lpage>3706</lpage><pub-id pub-id-type="doi">10.1097/JS9.0000000000001312</pub-id><pub-id pub-id-type="medline">38502861</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rao</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Hla</surname><given-names>M</given-names> </name><name name-style="western"><surname>Moor</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Multimodal generative AI for medical image interpretation</article-title><source>Nature</source><year>2025</year><month>03</month><volume>639</volume><issue>8056</issue><fpage>888</fpage><lpage>896</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-08675-y</pub-id><pub-id pub-id-type="medline">40140592</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Galbusera</surname><given-names>F</given-names> </name><name name-style="western"><surname>Casaroli</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bassani</surname><given-names>T</given-names> </name></person-group><article-title>Artificial intelligence and machine learning in spine research</article-title><source>JOR Spine</source><year>2019</year><month>03</month><volume>2</volume><issue>1</issue><fpage>e1044</fpage><pub-id pub-id-type="doi">10.1002/jsp2.1044</pub-id><pub-id pub-id-type="medline">31463458</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Youn</surname><given-names>BY</given-names> </name></person-group><article-title>Ethical implications of artificial intelligence in sport: a systematic scoping review</article-title><source>J Sport Health Sci</source><year>2025</year><month>12</month><volume>14</volume><fpage>101047</fpage><pub-id pub-id-type="doi">10.1016/j.jshs.2025.101047</pub-id><pub-id pub-id-type="medline">40316133</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Cui</surname><given-names>Z</given-names> </name></person-group><article-title>Impact of AI-assisted diagnosis on American patients&#x2019; trust in and intention to seek help from health care professionals: randomized, web-based survey experiment</article-title><source>J Med Internet Res</source><year>2025</year><month>06</month><day>18</day><volume>27</volume><fpage>e66083</fpage><pub-id pub-id-type="doi">10.2196/66083</pub-id><pub-id pub-id-type="medline">40532180</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Al-Sammarraie</surname><given-names>RN</given-names> </name><name name-style="western"><surname>Al Mubasher</surname><given-names>H</given-names> </name><name name-style="western"><surname>Awad</surname><given-names>M</given-names> </name><etal/></person-group><article-title>An artificial intelligence-aided scoping review of medicinal plant research in the Fertile Crescent</article-title><source>Front Pharmacol</source><year>2025</year><volume>16</volume><fpage>1542709</fpage><pub-id pub-id-type="doi">10.3389/fphar.2025.1542709</pub-id><pub-id pub-id-type="medline">40529497</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaya</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yavuz</surname><given-names>S</given-names> </name></person-group><article-title>Can generative AI and ChatGPT break human supremacy in mathematics and reshape competence in cognitive-demanding problem-solving tasks?</article-title><source>J Intell</source><year>2025</year><month>04</month><day>2</day><volume>13</volume><issue>4</issue><fpage>43</fpage><pub-id pub-id-type="doi">10.3390/jintelligence13040043</pub-id><pub-id pub-id-type="medline">40278052</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>R code for the statistical analyses of urology learning outcomes.</p><media xlink:href="jmir_v28i1e89315_app1.txt" xlink:title="TXT File, 4 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Self-reported frequency of generative AI use among participants (n=105).</p><media xlink:href="jmir_v28i1e89315_app2.png" xlink:title="PNG File, 209 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 1</label><p>CONSORT-EHEALTH checklist (V 1.6.1).</p><media xlink:href="jmir_v28i1e89315_app3.pdf" xlink:title="PDF File, 13326 KB"/></supplementary-material></app-group></back></article>