<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e86726</article-id><article-id pub-id-type="doi">10.2196/86726</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Comparison of Two AI Chatbots for Diagnosis and Providing Treatment Suggestions in Retinopathy of Prematurity: Retrospective Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Peng</surname><given-names>Shaojuan</given-names></name><degrees>MBBS</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Zhao</surname><given-names>Xinyu</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Wu</surname><given-names>Zhenquan</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yuan</surname><given-names>Duo</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Duan</surname><given-names>Na</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cui</surname><given-names>Kaixuan</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yu</surname><given-names>Zhen</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yang</surname><given-names>Weihua</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wei</surname><given-names>Wenbin</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chi</surname><given-names>Wei</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Zhang</surname><given-names>Guoming</given-names></name><degrees>MD, PHD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Jinan University</institution><addr-line>Guangzhou</addr-line><country>China</country></aff><aff id="aff2"><institution>Shenzhen Eye Medical Center, Shenzhen Eye Hospital, Southern Medical University</institution><addr-line>18 Zetian Road, Futian District</addr-line><addr-line>Shenzhen</addr-line><country>China</country></aff><aff id="aff3"><institution>Huizhou Third People&#x2019;s Hospital, First Affiliated Hospital of Jinan University</institution><addr-line>Guangzhou</addr-line><country>China</country></aff><aff id="aff4"><institution>Beijing Tongren Eye Center, Capital Medical University</institution><addr-line>Beijing</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Mansoor</surname><given-names>Masab</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Lee</surname><given-names>Seung Won</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Lang</surname><given-names>Stefan</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Guoming Zhang, MD, PHD, Shenzhen Eye Medical Center, Shenzhen Eye Hospital, Southern Medical University, 18 Zetian Road, Futian District, Shenzhen, 518043, China, 1 0755-23959600; <email>zhang-guoming@163.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>7</day><month>10</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e86726</elocation-id><history><date date-type="received"><day>29</day><month>10</month><year>2025</year></date><date date-type="rev-recd"><day>16</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>18</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Shaojuan Peng, Xinyu Zhao, Zhenquan Wu, Duo Yuan, Na Duan, Kaixuan Cui, Zhen Yu, Weihua Yang, Wenbin Wei, Wei Chi, Guoming Zhang. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 7.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e86726"/><abstract><sec><title>Background</title><p>Retinopathy of Prematurity (ROP) is a leading cause of preventable childhood blindness; yet, a global shortage of experienced pediatric ophthalmologists impedes timely diagnosis and treatment. While emerging AI chatbots are promising clinical decision-support tools in some ophthalmic diseases, their performance in ROP diagnosis and providing treatment suggestions remains uncertain.</p></sec><sec><title>Objective</title><p>This study aimed to compare the performance of Google&#x2019;s Gemini 2.5 Pro and OpenAI&#x2019;s ChatGPT o4-mini in ROP diagnosis and providing treatment suggestions against the gold standard of clinical consensus.</p></sec><sec sec-type="methods"><title>Methods</title><p>A retrospective analysis was conducted on 70 infants (140 eyes) with treatment-requiring ROP, each providing structured clinical text data and wide-field fundus images. We adopted a 2-stage prompting strategy for AI chatbots, instructing them first to generate ROP diagnoses (including zone, stage, and presence of plus disease), and subsequently to provide treatment suggestions. After collecting the generated responses, we assessed their performance by comparing the consistency of their diagnosis and treatment suggestions with the consensus gold standard. Furthermore, 2 independent specialists quantitatively assessed the outputs of Gemini 2.5 Pro and ChatGPT o4-mini using the ROP-specific Global Quality Score (GQS), which is a 5-point scale ranging from 1 (unusable) to 5 (excellent). Statistical significance was set at <italic>P</italic>&#x003C;.05, and all statistical analyses were performed using R (version 4.4.1; R Foundation for Statistical Computing).</p></sec><sec sec-type="results"><title>Results</title><p>For the tasks of ROP zoning and staging, the consistency rates between Gemini 2.5 Pro and ChatGPT o4-mini were 79.3% (111/140) vs 85.7% (120/140; zone), 64.3% (90/140) vs 70.0% (98/140; stage), respectively. For the task of treatment requirement, the rates (also referred to as sensitivity) were 93.6% (131/140) vs 90.7% (127/140), respectively. None of these differences were statistically significant (<italic>P</italic>&#x003E;.05). However, Gemini 2.5 Pro showed significantly better performance in plus disease identification (consistency: 108/140, 77.1% vs 80/140, 57.1%; <italic>P</italic>=.006), while ChatGPT o4-mini demonstrated significantly higher guideline adherence in treatment modality suggestions based on gold-standard ROP diagnoses (consistency: 91/140, 65.0%; vs 55/140, 39.3%; <italic>P</italic>=.01). Compared to Gemini 2.5 Pro, ChatGPT o4-mini performed better in providing ROP treatment suggestions (GQS score; <italic>P</italic>=.001), while the 2 AI chatbots had comparable GQS scores in diagnostic tasks.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>ChatGPT o4-mini shows greater promise in generating evidence-based treatment suggestions based on gold-standard diagnoses, whereas Gemini 2.5 Pro shows advantages in visual interpretation, supporting its potential for targeted ROP diagnostic screening, particularly in identifying plus disease. As these AI chatbots continue to evolve, their performance merits further validation using larger cohorts.</p></sec></abstract><kwd-group><kwd>Artificial intelligence chatbot</kwd><kwd>retinopathy of prematurity</kwd><kwd>wide-field fundus image</kwd><kwd>diagnosis</kwd><kwd>treatment suggestion</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Retinopathy of prematurity (ROP) is one of the leading causes of preventable childhood blindness worldwide, and its effective prevention and control have become an urgent global public health challenge [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Particularly in many middle-income countries [<xref ref-type="bibr" rid="ref3">3</xref>], the improvement of neonatal intensive care capabilities has significantly increased the survival rate of low-birth-weight infants [<xref ref-type="bibr" rid="ref2">2</xref>]. However, this increase has been accompanied by a rising incidence of ROP, which exacerbates the burden of childhood blindness [<xref ref-type="bibr" rid="ref4">4</xref>]. Given the rapidly growing demand for ROP screening and treatment, the global shortage of experienced pediatric ophthalmologists has become increasingly severe, particularly in resource-limited regions [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. In such regions, auxiliary diagnostic and treatment tools can reduce screening costs while supporting ophthalmologists in expanding service coverage through telemedicine [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. This has further driven an unprecedented need for scalable, accurate, and standardized auxiliary tools [<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>The clinical management of ROP strictly adheres to evidence-based guidelines, and its diagnosis and treatment suggestions depend highly on the precise interpretation of fundus vascular and lesion features [<xref ref-type="bibr" rid="ref10">10</xref>]. The third edition of the International Classification of ROP (ICROP3) updated and refined key indicators, including lesion zone, stage, and plus disease [<xref ref-type="bibr" rid="ref11">11</xref>]. Wide-field fundus imaging is crucial not only for accurate diagnosis but also for providing objective visual evidence. These include the extent and morphology of the avascular retinal area, which are the core basis for guiding interventions such as laser therapy or anti&#x2013;vascular endothelial growth factor (VEGF) injection [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. In recent years, efficient feature preservation segmentation networks have further optimized the boundary delineation of fundus structures and lesions, thereby improving the reliability of AI-driven image interpretation [<xref ref-type="bibr" rid="ref13">13</xref>]. To bridge the gap between these sophisticated diagnostic requirements and the global shortage of expert ophthalmologists, innovative auxiliary tools are urgently needed. Accordingly, AI chatbots that can integrate clinical text data and medical images may offer potential support in ROP diagnosis and providing treatment suggestions [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>In recent years, AI chatbots such as OpenAI&#x2019;s ChatGPT [<xref ref-type="bibr" rid="ref16">16</xref>], Google&#x2019;s Gemini [<xref ref-type="bibr" rid="ref17">17</xref>], and DeepSeek (Liang Wenfeng) [<xref ref-type="bibr" rid="ref18">18</xref>] have demonstrated tangible efficacy in diagnosis and providing treatment suggestions for ophthalmic diseases [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>]. For fundus disorders such as retinal detachment, age-related macular degeneration, and macular hole, both ChatGPT and Google Gemini exhibited consistent assessments of clinical records, showing high concordance with retinal specialists&#x2019; evaluations [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. Similarly, for other major fundus diseases, including age-related macular degeneration, glaucoma, and vitreous hemorrhage, ChatGPT showed preliminary diagnostic potential based solely on fundus photographs, although its performance remained variable [<xref ref-type="bibr" rid="ref24">24</xref>]. The uncertain efficacy of AI chatbots in ROP diagnosis and providing treatment suggestions warrants a comparative analysis of their performance using a text-image fusion approach.</p><p>To compare the performance of Gemini 2.5 Pro and ChatGPT o4-mini in ROP diagnosis and providing treatment suggestions, we collected 70 infants with treatment-requiring ROP and evaluated the performance of the two AI chatbots. This study provides key empirical evidence and improvement directions for the application, evaluation, and iterative optimization of the 2 AI chatbots in pediatric ophthalmology and other imaging-driven clinical specialties.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>This retrospective study was conducted at Shenzhen Eye Hospital (SZEH) and ultimately enrolled 70 infants (140 eyes in total) with ROP who underwent initial treatment for analysis in 2024. The entire process is illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study design and process for comparing the performance of two AI chatbots in the diagnosis and providing treatment suggestions for retinopathy of prematurity. The process consisted of 3 key steps: (A) medical documentation and specialist standards; (B) evaluation and prompts; (C) outcome analyses. GQS: Global Quality Score; ROP: retinopathy of prematurity.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e86726_fig01.png"/></fig></sec><sec id="s2-2"><title>Medical Documentation and Specialist Standards</title><p>We collected demographic and clinical data for each infant using a standardized structured template, including sex, gestational age (GA), birth weight (BW), pregnancy type, and mode of delivery [<xref ref-type="bibr" rid="ref25">25</xref>]. All structured fields were presented to both AI chatbots in an identical format to ensure fair comparisons. GA was recorded in weeks, and BW in grams, using a uniform numerical format. Clinical data comprised preoperative, deidentified bilateral fundus images obtained from ROP screening reports.</p><p>To ensure the reliability of the gold standard used for evaluating AI chatbot performance, we invited 3 senior pediatric retinal specialists (XZ, ZW, and GZ) to independently review the same ROP cases. Each specialist was provided with clinical text records and corresponding deidentified bilateral fundus images. Their tasks included (1) confirming the &#x201C;gold standard&#x201D; diagnosis for each case, which involved defining the ROP zone, stage, and presence of plus disease in accordance with the ICROP3 guidelines [<xref ref-type="bibr" rid="ref11">11</xref>]; (2) establishing the optimal treatment suggestion, including the requirement for treatment and the selection of treatment modality (laser therapy or anti-VEGF injection). Fleiss kappa coefficient was used to assess interrater reliability. If all 3 specialists (XZ, ZW, and GZ) reached a unanimous agreement on the diagnosis and treatment, this consensus was finalized as the gold standard. Any disagreements were resolved through group discussion until a consensus was reached.</p></sec><sec id="s2-3"><title>Selection of Infants With ROP</title><p>Inclusion criteria were applied to ensure data integrity and quality: (1) preterm infants with GA &#x003C;32 weeks or BW &#x003C;2000 g, (2) ROP diagnosed according to the ICROP3 guidelines[<xref ref-type="bibr" rid="ref11">11</xref>] and who underwent initial laser therapy or anti-VEGF injection, (3) complete clinical data available, including ROP screening reports (binocular ROP zone, stage, and plus disease assessment), preoperative wide-field fundus images, and treatment records.</p><p>Exclusion criteria were as follows: (1) incomplete medical history or clinical information, (2) absence of key preoperative or postoperative reports, (3) previous ROP treatment at other hospitals, (4) ROP stage &#x2265;4 (due to more complex treatment regimens and limited feasibility of standardized surgical intervention).</p></sec><sec id="s2-4"><title>Evaluation and Prompt</title><sec id="s2-4-1"><title>Selection of AI Chatbots</title><p>The rapid expansion of AI chatbot research in health care [<xref ref-type="bibr" rid="ref26">26</xref>-<xref ref-type="bibr" rid="ref28">28</xref>] has witnessed the emergence of mainstream models such as OpenAI&#x2019;s ChatGPT [<xref ref-type="bibr" rid="ref16">16</xref>], Google&#x2019;s Gemini [<xref ref-type="bibr" rid="ref17">17</xref>], and DeepSeek [<xref ref-type="bibr" rid="ref18">18</xref>], along with their subversions. Early versions such as ChatGPT-3.5 only supported text interaction and were unable to process medical images [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. Among available Gemini subversions, we selected Gemini 2.5 Pro because complex ROP assessment requires detailed ophthalmic image interpretation and task-specific validation of multimodal reasoning in ophthalmology [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. Specifically, Gemini 2.5 Pro supports text and image inputs, and provides a long context window, making it suitable for processing structured clinical records and fundus images in this text-image fusion task [<xref ref-type="bibr" rid="ref33">33</xref>]. The development of medical vision-language models and foundation models in ophthalmology further supports the potential value of multimodal AI systems for ophthalmic image interpretation, while also highlighting the need for task-specific validation [<xref ref-type="bibr" rid="ref34">34</xref>]. For ChatGPT o4-mini, the HealthBench framework released by OpenAI in 2025 provides a relevant reference for evaluating clinical reasoning and safety in medical AI systems [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. OpenAI documentation indicates that ChatGPT o4-mini supports text and image inputs and is optimized for fast and effective reasoning, including visual tasks [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>]. This design is aligned with the technical challenge of simultaneously interpreting clinical records and wide-field fundus images in ROP diagnosis and providing treatment suggestions [<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>Therefore, our study selected Gemini 2.5 Pro and ChatGPT o4-mini for evaluation, as these AI chatbots meet the rigorous requirements of clinical tasks for multimodal data integration. Gemini 2.5 Pro was publicly available through Google&#x2019;s official platform during the study period [<xref ref-type="bibr" rid="ref33">33</xref>]. ChatGPT o4-mini used in this study refers to the reasoning-focused o4-mini model released by OpenAI in April 2025 [<xref ref-type="bibr" rid="ref37">37</xref>], distinct from the earlier ChatGPT-4o mini omni model released in 2024. We used its web-exclusive high-inference mode, an enhanced reasoning variant optimized for complex multimodal and clinical reasoning tasks. All chatbot interactions in this study were conducted exclusively via the official web interfaces of the respective providers, without reliance on API access. These standard web interfaces do not provide end-users with adjustable parameters such as temperature. The temperature value is fixed at a default setting determined by the platform providers and cannot be modified by users. Consequently, all AI chatbot outputs were generated under the default, fixed configuration of the web platforms to ensure consistency and reproducibility.</p></sec><sec id="s2-4-2"><title>AI Chatbot-Aided Diagnosis and Treatment Suggestions</title><p>We conducted standardized AI chatbot interactions with Gemini 2.5 Pro and ChatGPT o4-mini between June 1 and June 30, 2025. To reduce evasive responses that AI chatbots may generate when directly asked for medical advice (eg, stating &#x201C;I am not a medical professional and cannot provide treatment suggestions&#x201D;) [<xref ref-type="bibr" rid="ref22">22</xref>], this study adopted a 2-stage, specialist-simulation prompting strategy. The initial prompt framed the task as a &#x201C;clinical simulation analysis scenario&#x201D; and assigned the AI chatbots the role of an &#x201C;experienced pediatric fundus disease specialist&#x201D; with additional context: &#x201C;This is a clinical simulation analysis scenario. You will act as a pediatric ophthalmologist with rich clinical experience, and based on the basic medical history and a varying number of Retcam fundus images provided, briefly analyze the ROP zone, stage, and whether there is plus disease in this infant (report separately for left and right eyes, omitting the analysis process and directly providing the results).&#x201D;</p><p>In the diagnosis task, the 2 AI chatbots were provided with standardized clinical records and deidentified bilateral fundus images (consistent with those obtained by experts) for each infant. The left and right fundus images were uploaded as separate files (1600&#x00D7;1200 resolution) within this conversation, rather than as composite images with white dividing bars, thus minimizing potential interpretation errors related to image boundaries. Simultaneously, the two AI chatbots were asked: &#x201C;As a pediatric ophthalmologist with rich clinical experience, based on the provided patient history and Retcam fundus images provided above, please directly report the ROP zone, stage, and presence of plus disease in this infant&#x201D;</p><p>In the treatment suggestion task, we manually provided the gold-standard diagnoses (not the chatbot&#x2019;s own outputs) to the AI chatbots, with the following instruction: &#x201C;Based on my revised diagnoses of this infant, does the child require surgical intervention? If so, what is your recommended optimal surgical plan? Please briefly explain the rationale behind your suggestion&#x201D;</p><p>For each infant, the diagnosis and treatment suggestion tasks were conducted within the same continuous chat session. For different infants, separate independent chat sessions were used with no information carryover between cases. A single query was submitted per case, and only the initial response was included for analysis. All model hyperparameters (eg, temperature and top-p sampling) were retained at the default values of the official web interface, with no custom parameter adjustments. Screenshots of the response interfaces of the two AI chatbots are shown in <xref ref-type="fig" rid="figure2">Figures 2</xref> and <xref ref-type="fig" rid="figure3">3</xref>. All prompts and interactions were conducted in Chinese and that the English prompts shown in the figures are translations. This design separated the AI chatbots&#x2019; performance in providing evidence-based treatment suggestions from diagnostic errors, enabling an unbiased evaluation based on the gold-standard diagnoses. At the end of each AI chatbot interaction, 2 specialists (XZ and ZW) independently graded the quality of the AI chatbots&#x2019; outputs using an ROP-specific Global Quality Score (GQS) adapted from Bernard et al [<xref ref-type="bibr" rid="ref39">39</xref>].</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Screenshot of Gemini 2.5 Pro&#x2019;s response interface for an infant with retinopathy of prematurity.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e86726_fig02.png"/></fig><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Screenshot of ChatGPT o4-mini&#x2019;s response interface for an infant with retinopathy of prematurity.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e86726_fig03.png"/></fig><p>The primary outcome of this study was to explore the performance of Gemini 2.5 Pro and ChatGPT o4-mini in analyzing infants with ROP and generating coherent, accurate diagnoses and treatment suggestions by comparing their outputs to the specialists&#x2019; consensus gold standard (including consistency rates of diagnoses and treatment suggestions). For treatment modality selection, a response that selects an effective treatment modality consistent with gold-standard decision criteria was rated as correct. As a secondary outcome, we used the mean GQS to quantify the subjective quality and clinical utility of the 2 AI chatbots&#x2019; output results.</p></sec><sec id="s2-4-3"><title>ROP-Specific GQS</title><p>As shown in <xref ref-type="table" rid="table1">Table 1</xref>, the scoring system adopted a 5-point Likert scale (1=unusable, 5=excellent).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Retinopathy of prematurity specific Global Quality Score criteria.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Score</td><td align="left" valign="bottom">Diagnosis</td><td align="left" valign="bottom">Treatment suggestion</td></tr></thead><tbody><tr><td align="left" valign="top">5 (Excellent)</td><td align="left" valign="top">All bilateral diagnoses fully consistent with the gold standard</td><td align="left" valign="top">The treatment suggestions consistent with the gold standard, integrate individual characteristics and ICROP3<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> guidelines, mention posttreatment monitoring, no hallucinations</td></tr><tr><td align="left" valign="top">4<break/>(Good)</td><td align="left" valign="top">5 diagnoses consistent with the gold standard</td><td align="left" valign="top">The treatment suggestions consistent with the gold standard, integrate ICROP3 guidelines, lack of individualized details or posttreatment monitoring, no hallucinations</td></tr><tr><td align="left" valign="top">3 (Moderate)</td><td align="left" valign="top">3&#x2010;4 diagnoses consistent with the gold standard</td><td align="left" valign="top">The treatment suggestions inconsistent with the gold standard, integrate ICROP3 guidelines, no hallucinations</td></tr><tr><td align="left" valign="top">2<break/>(Poor)</td><td align="left" valign="top">1&#x2010;2 diagnoses consistent with the gold standard</td><td align="left" valign="top">The treatment suggestions inconsistent with the gold standard, not integrate ICROP3 guidelines, hallucinations exist</td></tr><tr><td align="left" valign="top">1 (Unusable)</td><td align="left" valign="top">No diagnosis consistent with the gold standard or no response provided</td><td align="left" valign="top">The treatment suggestions contain fundamental errors (eg, no treatment required) or no suggestion is provided, hallucinations exist</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>ICROP3: third edition of the International Classification of retinopathy of prematurity.</p></fn></table-wrap-foot></table-wrap><p>The ROP-specific GQS was adapted from the tool developed by Bernard et al [<xref ref-type="bibr" rid="ref39">39</xref>], originally designed for website usability assessment. Given the lack of standardized quantitative instruments for evaluating clinical AI chatbots&#x2019; outputs, the GQS has been adopted in ophthalmic AI studies [<xref ref-type="bibr" rid="ref40">40</xref>] and digital health research [<xref ref-type="bibr" rid="ref41">41</xref>], supporting its use in our investigation. To better assess the quality of AI chatbot outputs in real ROP case analyses, we modified the GQS according to the ICROP3 guidelines [<xref ref-type="bibr" rid="ref11">11</xref>] to generate ROP-specific GQS. We also incorporated hallucination assessment for treatment suggestion outputs [<xref ref-type="bibr" rid="ref42">42</xref>]. Prior to scoring, 2 ROP specialist raters (XZ and ZW) underwent standardized training covering ICROP3 criteria and ROP-specific GQS scoring rules, with pilot calibration and discussion of borderline cases. To minimize bias, AI chatbots&#x2019; outputs were deidentified and provided as plain text, ensuring raters were blinded to the source of AI chatbots. Interrater reliability for assessing AI chatbots was evaluated using the intraclass correlation coefficient (ICC).</p></sec></sec><sec id="s2-5"><title>Statistical Analysis</title><p>All statistical analyses were performed using R (version 4.4.1; R Foundation for Statistical Computing). The &#x201C;gold standard&#x201D; was established by the consensus of 3 pediatric retinal specialists (XZ, ZW, and GZ). Interrater reliability regarding ROP zone, stage, plus disease, treatment requirement, and treatment modality were assessed using the Fleiss kappa coefficient. Any disagreements among the 3 pediatric retinal specialists were resolved through discussion until a consensus was reached. Generalized estimating equations (GEE) were used to compare the consistency rates and GQS of the 2 models while accounting for intereye correlation. As all enrolled eyes in this study were treatment-requiring ROP, the metric for treatment requirement was equivalent to sensitivity. Subgroup analyses of the 2 AI chatbots for diagnosis and providing treatment suggestions were conducted between Zone I and Zone II (as shown in <xref ref-type="table" rid="table2">Table 2</xref>), because Zone III cases were insufficient for statistical analysis. For the 3 Zone III eyes, both ChatGPT o4-mini and Gemini 2.5 Pro misclassified all 3 eyes as Zone II. For the Zone I subgroup, complete separation was observed for Gemini 2.5 Pro in both the treatment requirement and treatment modality comparisons, as it achieved 100% correct classification. This separation precludes the use of conventional GEE Wald-type estimates due to their unreliability. Consequently, for these 2 comparisons, we applied a penalized GEE approach incorporating Firth likelihood penalty, which provides reliable estimates of the regression coefficients while properly accounting for the inherent correlation between eyes. For the GQS, each AI chatbot had 140 individual rater scores for each task. The mean GQS for each response was calculated by averaging the scores from the two raters, followed by the overall mean score. Score category frequencies and hallucination rates were calculated using the original integer scores provided by both raters. <italic>P</italic>&#x003C;.05 was considered statistically significant.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Subgroup analysis of two AI chatbots for diagnosis and providing treatment suggestions in Zones I and II.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome</td><td align="left" valign="bottom" colspan="4">Zone I<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> (n=15)</td><td align="left" valign="bottom" colspan="4">Zone II<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> (n=122)</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">Gemini 2.5 Pro, n (%)</td><td align="left" valign="top">ChatGPT o4-mini, n (%)</td><td align="left" valign="top">OR<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup><break/>(95% CI)</td><td align="left" valign="top"><italic>P</italic> value</td><td align="left" valign="top">Gemini 2.5 Pro, n (%)</td><td align="left" valign="top">ChatGPT o4-mini, n (%)</td><td align="left" valign="top">OR<break/>(95% CI)</td><td align="left" valign="top"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Zone identification</td><td align="left" valign="top">10 (66.7)</td><td align="left" valign="top">2 (13.3)</td><td align="left" valign="top">0.08 (0.01-0.63)</td><td align="left" valign="top">.02</td><td align="left" valign="top">100 (82.0)</td><td align="left" valign="top">118 (96.7)</td><td align="left" valign="top">6.49<break/>(1.53-27.55)</td><td align="left" valign="top">.01</td></tr><tr><td align="left" valign="top">Stage</td><td align="left" valign="top">9 (60.0)</td><td align="left" valign="top">12 (80.0)</td><td align="left" valign="top">2.70<break/>(0.69-10.64)</td><td align="left" valign="top">.16</td><td align="left" valign="top">80 (65.6)</td><td align="left" valign="top">84 (68.9)</td><td align="left" valign="top">1.16<break/>(0.60-2.23)</td><td align="left" valign="top">.66</td></tr><tr><td align="left" valign="top">Plus disease</td><td align="left" valign="top">13 (86.7)</td><td align="left" valign="top">13 (86.7)</td><td align="left" valign="top">1.00<break/>(0.04-26.23)</td><td align="left" valign="top">1.0</td><td align="left" valign="top">94 (77.0)</td><td align="left" valign="top">65 (53.3)</td><td align="left" valign="top">0.34<break/>(0.17-0.68)</td><td align="left" valign="top">.003</td></tr><tr><td align="left" valign="top">Treatment requirement</td><td align="left" valign="top">15 (100)</td><td align="left" valign="top">14 (93.3)</td><td align="left" valign="top">0.18<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup><break/>(0.02-1.54)</td><td align="left" valign="top">.12<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">113 (92.6)</td><td align="left" valign="top">110 (90.2)</td><td align="left" valign="top">0.73<break/>(0.45-1.17)</td><td align="left" valign="top">.19</td></tr><tr><td align="left" valign="top">Treatment modality</td><td align="left" valign="top">15 (100)</td><td align="left" valign="top">5 (33.3)</td><td align="left" valign="top">0.01<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup><break/>(0.002-0.05)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">40 (32.8)</td><td align="left" valign="top">83 (68.0)</td><td align="left" valign="top">4.36<break/>(1.75-10.88)</td><td align="left" valign="top">.002</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>For Zone I, no significant differences were observed in most outcomes, with the exception of ROP zone identification and treatment modality.</p></fn><fn id="table2fn2"><p><sup>b</sup>For Zone II, ChatGPT o4-mini significantly outperformed Gemini 2.5 Pro in ROP zone identification (OR 6.49; <italic>P</italic>=.011) and treatment modality selection (OR 4.36; <italic>P</italic>=.002). Conversely, ChatGPT o4-mini showed significantly lower consistency in plus disease identification compared with Gemini 2.5 Pro (OR 0.34; <italic>P</italic>=.003). <italic>P</italic>&#x003C;.05 was considered statistically significant.</p></fn><fn id="table2fn3"><p><sup>c</sup>OR: odds ratio.</p></fn><fn id="table2fn4"><p><sup>d</sup>For the Zone I subgroup, complete separation occurred for Gemini 2.5 Pro in the treatment requirement and treatment modality comparisons (100% correct). Accordingly, a Firth-penalized GEE was applied to address this separation. Other statistical comparisons were performed using GEE. All ORs reflect ChatGPT o4-mini versus Gemini 2.5 Pro. </p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-6"><title>Ethical Considerations</title><p>The study was approved by the Institutional Review Board (IRB) of SZEH (IRB No. 2025KYPJ120) prior to its initiation. Informed consent was waived by the IRB due to the retrospective study design and the use of fully deidentified data for secondary analysis. To ensure compliance with ethical standards and data privacy regulations, all sensitive personal information was removed from case descriptions, only ROP-related key factors such as GA and BW were retained to balance research accuracy and privacy protection. All fundus images were anonymized, free of any potential biometric identifiers or embedded textual metadata, and subjected to quality control in adherence to data privacy protocols. This study did not recruit any additional subjects and no compensation was provided to participants.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview</title><p>Demographic characteristics of the 70 infants are shown in <xref ref-type="table" rid="table3">Table 3</xref>. Specifically, the mean GA was 26.79 (SD 2.14) weeks, and the mean BW was 888.60 (SD 277.82) g. Among all infants, 57.14% (40/70) were male; 64.29% (45/70) were singleton pregnancies; and the proportion of vaginal deliveries was comparable to that of cesarean sections.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Demographic characteristics of infants with retinopathy of prematurity (N=70).</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristics</td><td align="left" valign="bottom">Values</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Sex, n (%)</td></tr><tr><td align="left" valign="top">&#x2003;Male</td><td align="left" valign="top">40 (57.14)</td></tr><tr><td align="left" valign="top">&#x2003;Female</td><td align="left" valign="top">30 (42.86)</td></tr><tr><td align="left" valign="top">Gestational age (weeks), mean (SD)</td><td align="left" valign="top">26.79 (2.14)</td></tr><tr><td align="left" valign="top">Birth weight (g), mean (SD)</td><td align="left" valign="top">888.60 (277.82)</td></tr><tr><td align="left" valign="top" colspan="2">Pregnancy type, n (%)</td></tr><tr><td align="left" valign="top">&#x2003;Singleton</td><td align="left" valign="top">45 (64.29)</td></tr><tr><td align="left" valign="top">&#x2003;Multiple pregnancy</td><td align="left" valign="top">25 (35.71)</td></tr><tr><td align="left" valign="top" colspan="2">Delivery type, n (%)</td></tr><tr><td align="left" valign="top">&#x2003;Vaginal delivery</td><td align="left" valign="top">37 (52.86)</td></tr><tr><td align="left" valign="top">&#x2003;Cesarean section</td><td align="left" valign="top">33 (47.14)</td></tr></tbody></table></table-wrap><p>Diagnosis and treatment suggestions for the 140 eyes (70 infants) are shown in <xref ref-type="table" rid="table4">Table 4</xref>. The most common diagnosis was Zone II, Stage 3 ROP. Specifically, Zone II involvement was observed in 62 of 70 left eyes (88.6%) and 60 of 70 right eyes (85.7%). Stage 3 was noted in 63 of 70 left eyes (90.0%) and 64 of 70 right eyes (91.4%). Furthermore, plus disease was highly prevalent, affecting 64 of 70 eyes (91.4%) bilaterally. All eyes required surgical treatment. Laser therapy was the primary modality, administered to 50 of 70 left eyes (71.4%) and 48 of 70 right eyes (68.6%). The remaining eyes were treated with anti-VEGF injections.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Baseline characteristics of infants with retinopathy of prematurity (ROP; N=70).</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" colspan="2">Characteristic and category</td><td align="left" valign="bottom">Left eye, n (%)</td><td align="left" valign="bottom">Right eye, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">ROP Zone</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zone I</td><td align="left" valign="top">7 (10.0)</td><td align="left" valign="top">8 (11.4)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zone II</td><td align="left" valign="top">62 (88.6)</td><td align="left" valign="top">60 (85.7)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zone III</td><td align="left" valign="top">1 (1.4)</td><td align="left" valign="top">2 (2.9)</td></tr><tr><td align="left" valign="top" colspan="4">ROP Stage</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Stage 1</td><td align="left" valign="top">1 (1.4)</td><td align="left" valign="top">1 (1.4)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Stage 2</td><td align="left" valign="top">6 (8.6)</td><td align="left" valign="top">5 (7.1)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Stage 3</td><td align="left" valign="top">63 (90.0)</td><td align="left" valign="top">64 (91.4)</td></tr><tr><td align="left" valign="top" colspan="4">Plus disease</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Absent (-)</td><td align="left" valign="top">6 (8.6)</td><td align="left" valign="top">6 (8.6)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Present (+)</td><td align="left" valign="top">64 (91.4)</td><td align="left" valign="top">64 (91.4)</td></tr><tr><td align="left" valign="top" colspan="4">Treatment requirement</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Yes</td><td align="left" valign="top">70 (100.0)</td><td align="left" valign="top">70 (100.0)</td></tr><tr><td align="left" valign="top" colspan="4">Treatment modality</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Laser therapy</td><td align="left" valign="top">50 (71.4)</td><td align="left" valign="top">48 (68.6)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>IVI<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">20 (28.6)</td><td align="left" valign="top">22 (31.4)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>IVI: intravitreal injection.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Interrater Reliability</title><p>Interrater reliability among the 3 specialists for the gold standard was substantial to excellent across all 5 tasks: Fleiss kappa for ROP zone=0.65 (95% CI 0.36&#x2010;0.82), for ROP stage=0.72 (95% CI 0.43&#x2010;0.90), for plus disease=0.66 (95% CI 0.34&#x2010;0.86), and for treatment modality=0.82 (95% CI 0.68&#x2010;0.92). For treatment requirement, all 3 specialists (XZ, ZW, and GZ) reached unanimous consensus, resulting in no calculated kappa value for this endpoint. Initial agreement was achieved in 84.3% (59/70) of cases; the remaining 11 cases with disagreement were resolved through discussion until full consensus was reached for all tasks. Subsequently, the consistency of GQS scoring for two AI chatbots&#x2019; outputs was evaluated by two independent ROP specialists (XZ and ZW) raters to ensure reliability of the secondary outcome measures. For Gemini 2.5 Pro, excellent interrater reliability was observed for both diagnostic GQS consistency (ICC=0.85; 95% CI 0.77&#x2010;0.90) and treatment suggestion GQS consistency (ICC=0.81; 95% CI 0.71&#x2010;0.88). Similarly, ChatGPT o4-mini demonstrated strong interrater reliability for diagnostic GQS consistency (ICC=0.86; 95% CI 0.78&#x2010;0.91) and treatment suggestion GQS consistency (ICC=0.83; 95% CI 0.66&#x2010;0.91).</p></sec><sec id="s3-3"><title>Overall Performance Comparison of Two AI Chatbots</title><p>We compared the consistency between Gemini 2.5 Pro and ChatGPT o4-mini in 5 core clinical tasks (ROP zone, stage, plus disease identification, treatment requirement, and treatment modality). As shown in <xref ref-type="fig" rid="figure4">Figure 4</xref>, Gemini 2.5 Pro and ChatGPT o4-mini demonstrated comparable consistency in ROP zoning (79.3% vs 85.7%), staging (64.3% vs 70.0%), and treatment requirement (93.6% vs 90.7%), with no statistically significant differences (all <italic>P</italic>&#x003E;.05). However, Gemini 2.5 Pro demonstrated significantly better consistency in diagnosing plus disease (77.1% vs 57.1%; <italic>P</italic>=.006), suggesting its advantage in visual pattern recognition tasks. In contrast, ChatGPT o4-mini showed significantly better consistency in treatment modality (65.0% vs 39.3%; <italic>P</italic>=.01), indicating strengths in providing treatment suggestions based on the gold standard diagnoses. As shown in <xref ref-type="table" rid="table2">Table 2</xref>, subgroup analyses revealed distinct strengths and limitations between the two AI chatbots across ROP zones. All odds ratios (ORs) reflect ChatGPT o4-mini versus Gemini 2.5 Pro. In Zone I, Gemini 2.5 Pro maintained superior consistency, particularly in treatment modality task (100.0% vs 33.3%; OR=0.01; <italic>P</italic>&#x003C;.001). Conversely, in Zone II, ChatGPT o4-mini significantly outperformed Gemini 2.5 Pro in ROP zone identification (96.7% vs 82.0%; OR=6.49; <italic>P</italic>=.01) and treatment modality selection (68.0% vs 32.8%; OR=4.36; <italic>P</italic>=.002). In addition, ChatGPT o4-mini showed significantly lower consistency in plus disease detection within Zone II compared with Gemini 2.5 Pro (53.3% vs 77.0%; OR=0.34; <italic>P</italic>=.003).</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Comparison of consistency between Gemini 2.5 Pro and ChatGPT o4-mini in diagnosis and providing treatment suggestions for retinopathy of prematurity. The Y-axis represents consistency (%). The X-axis denotes specific tasks (ROP zoning, staging, plus disease identification, treatment requirement, treatment modality). Ns: not significant; *<italic>:P</italic>&#x003C;.05, significant difference; **: <italic>P</italic>&#x003C;.01, significant difference. Gemini 2.5 Pro outperformed ChatGPT o4-mini in plus disease identification (<italic>P</italic>=.006), whereas ChatGPT o4-mini was superior in treatment modality selection (<italic>P</italic>=.01).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e86726_fig04.png"/></fig></sec><sec id="s3-4"><title>Subgroup Analyses of Two AI Chatbots According to ROP Zone</title><p>We conducted subgroup analyses to explore the two AI chatbots&#x2019; performance across distinct ROP risk zones, including Zone I (high-risk) and Zone II (intermediate-risk), to identify potential strengths and limitations within specific populations.</p><p>As shown in <xref ref-type="table" rid="table2">Table 2</xref>, in Zone I (n=15), ChatGPT o4-mini achieved numerically lower consistency rates than Gemini 2.5 Pro in zone identification (13.3% vs 66.7%; OR=0.08; <italic>P</italic>=.02), and treatment modality (33.3% vs 100%; OR=0.01; <italic>P</italic>&#x003C;.001). No significant differences were observed for ROP staging (<italic>P</italic>=.16), plus disease identification (<italic>P</italic>&#x003E;.99), and treatment requirement (<italic>P</italic>=.12) in Zone I.</p><p>In Zone II (n=122), ChatGPT o4-mini exhibited clear advantages in clinical decision-making. It significantly outperformed Gemini 2.5 Pro in both ROP zone identification (96.7% vs 82.0%; OR=6.49; <italic>P</italic>=.01) and treatment modality selection (68.0% vs 32.8%; OR=4.36; <italic>P</italic>=.002). Conversely, ChatGPT o4-mini showed significantly lower consistency in plus disease identification compared with Gemini 2.5 Pro (53.3% vs 77.0%; OR=0.34; <italic>P</italic>=.003). For ROP staging and treatment requirement, no statistically significant differences were observed between the two AI chatbots in Zone II.</p></sec><sec id="s3-5"><title>GQS Results Analysis</title><p>The GQS results are presented in <xref ref-type="fig" rid="figure5">Figure 5</xref>. Based on the content evaluated by the GQS, hallucinations in treatment suggestion were identified in ratings of 1 or 2. For the diagnosis task, Gemini 2.5 Pro obtained a higher mean GQS score (3.61; SD 1.42) than ChatGPT o4-mini (3.41; SD 1.45), with no statistically significant difference (<italic>P</italic>=.39). Notably, the similar SDs across both AI chatbots suggest comparable variability in diagnostic output quality. Gemini 2.5 Pro had score 1 ratings in 7.14% (10/140) and score 2 ratings in 12.14% (17/140), while ChatGPT o4-mini yielded score 1 ratings in 7.86% (11/140) and score 2 ratings in 15.71% (22/140). In terms of treatment suggestion task, ChatGPT o4-mini significantly outperformed Gemini 2.5 Pro (mean GQS: 3.97; SD 1.26 vs 3.29; SD 1.19; <italic>P</italic>=.001). The small difference in SDs suggests similar variability in treatment suggestion output quality. Gemini 2.5 Pro had score 1 ratings in 7.86% (11/140) and score 2 ratings in 14.29% (20/140), corresponding to a rating-level hallucination rate of 22.14% (31/140), while ChatGPT o4-mini yielded score 1 ratings in 7.14% (10/140) and score 2 ratings in 7.14% (10/140), corresponding to a rating-level hallucination rate of 14.29% (20/140). Hallucinations may manifest as proposing ineffective treatment modalities not endorsed by the ICROP3 guidelines, exaggerating the indications for a particular intervention, or fabricating unproven reasons for a treatment modality preference. These results showed that ChatGPT o4-mini achieved a significantly higher mean GQS score than Gemini 2.5 Pro for treatment suggestions and received fewer hallucination-related ratings, supporting its superior performance in providing clinical ROP treatment suggestions in this study.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Comparison of global quality scores (GQS) for diagnosis and providing treatment suggestion in retinopathy of prematurity between Gemini 2.5 Pro and ChatGPT o4-mini. The bar chart displays the mean GQS for both (A) Diagnosis and (B) Treatment suggestion. For diagnostic task, the average GQS was 3.61 for Gemini 2.5 Pro vs 3.41 for ChatGPT o4-mini (<italic>P</italic>=.39). For treatment suggestion task, the average GQS was 3.29 for Gemini 2.5 Pro vs 3.97 for ChatGPT o4-mini (<italic>P</italic>=.001). Error bars represent SD. ns: not statistically significant difference; **(<italic>P</italic>&#x003C;.01 (significant difference).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e86726_fig05.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings and Comparison to Prior Work</title><p>This study compared the performance of Gemini 2.5 Pro and ChatGPT o4-mini for diagnosing and providing treatment suggestions in ROP. The results provided key empirical evidence for the performance of AI chatbots to bridge the gap between the complex diagnostic requirements of ROP and the shortage of pediatric ophthalmologists. Furthermore, our findings may offer directions for improvement, with potential implications for the wider application and evaluation of AI chatbots in other imaging-driven specialties, consistent with the broader translational framework of AI in smart health care [<xref ref-type="bibr" rid="ref43">43</xref>].</p><p>The tasks of ROP zoning, staging, and treatment requirement are relatively structured because they are based on standardized disease classification and treatment decision criteria provided by the ICROP3 guidelines [<xref ref-type="bibr" rid="ref11">11</xref>]. This guideline helps AI chatbots transform complex clinical observation tasks into highly standardized and repeatable operational procedures, thus demonstrating consistently reliable performance in these tasks [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. Both AI chatbots were provided with the gold-standard diagnoses (zone, stage, and presence of plus disease) before answering the question &#x201C;Is surgical treatment required?,&#x201D; as this task was highly objective and did not require complex clinical reasoning. Based on these standardized features, the 2 AI chatbots in this study showed relatively consistent performance on these tasks by leveraging their multimodal processing capabilities and the standardized criteria provided by the ICROP3 guidelines [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref45">45</xref>].</p><p>The identification of plus disease in ROP requires precise visual discrimination of morphological changes in retinal vasculature [<xref ref-type="bibr" rid="ref46">46</xref>]. Gemini 2.5 Pro exhibited superior consistency in identifying plus disease in this study. This advantage may be related to the broader ability of modern vision-language models to integrate visual features through multimodal representation learning and adaptive feature integration mechanisms [<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>]. In contrast, although ChatGPT o4-mini supports text-image reasoning and visual tasks according to OpenAI documentation [<xref ref-type="bibr" rid="ref49">49</xref>], its lower consistency in plus disease identification in this study suggests that general-purpose multimodal reasoning may not fully translate into optimal recognition of subtle ROP vascular abnormalities. This performance difference appears to be task-dependent. Prior ophthalmology studies have shown that the relative performance of AI chatbots varies across examination-based and clinical ophthalmology tasks [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>]. This finding does not contradict our results but rather highlights the &#x201C;performance specialization&#x201D; of AI chatbots [<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>]. These differences may be associated with model-specific optimization strategies, training data composition, input format, and task types [<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. Specifically, the training data varied: our study used wide-field fundus images, whereas the exam study used typical test-related images. Furthermore, these tasks impose different cognitive demands: our study focused on high-fidelity visual discrimination, whereas exam-based studies require textual integration of visual materials of diverse complexity. Therefore, vision-optimized AI chatbots such as Gemini 2.5 Pro are more adaptable for clinical tasks that rely on fine visual pattern recognition, as their performance aligns closely with the visual dependency of the task.</p><p>In tasks of determining treatment modality, which demand the integration of multidimensional clinical information (including clinical text records and wide-field fundus images), AI chatbots need to simulate specialists&#x2019; multifactor trade-off logic. Consistent with prior ophthalmic multimodal AI chatbot research [<xref ref-type="bibr" rid="ref21">21</xref>], ChatGPT o4-mini may have benefited from integrating structured text records and fundus images when deducing the optimal treatment modality, yielding superior performance. Previous medical large language model studies also suggest that language models can perform well on structured medical knowledge and reasoning tasks [<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>]. For questions based on established medical knowledge with clear and objective answers, these AI chatbots achieved strong performance. This indicates that when tasks are governed by explicit rules and objective criteria, the performance gap among different AI chatbot architectures narrows, and their capabilities converge.</p><p>Subgroup analysis showed that in high-risk Zone I, characterized by severe posterior retinal lesions and vascular abnormalities [<xref ref-type="bibr" rid="ref11">11</xref>], Gemini 2.5 Pro outperformed ChatGPT o4-mini in ROP zone identification and treatment modality selection. This finding may again reflect Gemini 2.5 Pro&#x2019;s stronger task-specific visual discrimination in cases requiring fine vascular pattern recognition, suggesting its potential value in the identification of complex retinal vascular diseases. Conversely, in the intermediate-risk Zone II with relatively mild vascular lesions, ChatGPT o4-mini significantly outperformed Gemini 2.5 Pro in both zone identification and treatment modality selection. Since treatment decisions in Zone II rely more heavily on the comprehensive clinical context rather than isolated visual examination results [<xref ref-type="bibr" rid="ref11">11</xref>], ChatGPT o4-mini may have benefited from integrating structured clinical records and fundus images when generating treatment suggestions, consistent with prior ophthalmic multimodal AI chatbot research [<xref ref-type="bibr" rid="ref21">21</xref>]. It is worth noting that ChatGPT o4-mini&#x2019;s ability to identify plus disease is not as good as Gemini 2.5 Pro, which is consistent with the overall comparison results. The above results confirm that the performance of the AI chatbot is closely related to the risk of the ROP zone. Gemini 2.5 Pro is more suitable for high-risk Zone I cases that require precise visual identification, while ChatGPT o4-mini is more suitable for medium-risk Zone II cases. This also verifies the overall performance of the two AI chatbots in different clinical scenarios and their robustness in handling different retinal complexities.</p><p>The difference in GQS between ChatGPT o4-mini and Gemini 2.5 Pro varies across task dimensions, reflecting the interplay between specialist scoring logic and AI chatbot performance. According to our ROP-specific GQS criteria, specialists mainly evaluated completeness and clinical practicality. When scoring, specialists particularly focus on whether the AI chatbots&#x2019; output fully covers the core diagnostic elements of ROP cases and whether it conforms to the follow-up or treatment suggestions. In the diagnosis dimension, no significant difference in GQS was observed between the two AI chatbots, indicating comparable performance in capturing ROP diagnostic details. In contrast, ChatGPT o4-mini achieved a significantly higher GQS in the treatment suggestion dimension. This advantage may be partly explained by its structured output format, which is compatible with clinical workflows and may help specialists extract key information efficiently [<xref ref-type="bibr" rid="ref55">55</xref>]. However, because the training data and optimization details of commercial AI chatbots are not fully disclosed, this interpretation should be regarded as a task-specific observation rather than a confirmed model-level mechanism. Although Gemini 2.5 Pro performs well in isolated image analysis, it often experiences logical discontinuities when integrating multidimensional textual information of ROP cases, resulting in a decreased matching degree between the output and actual clinical requirements.</p><p>Notably, the SD patterns differed between the 2 AI chatbots. For Gemini 2.5 Pro, the SDs for diagnostic GQS were higher than that for treatment suggestion GQS, while ChatGPT o4-mini exhibited higher SDs in treatment suggestion GQS. This aligns with the common observation of insufficient output consistency in AI chatbots for complex clinical decisions, particularly evident in multimodal information integration tasks [<xref ref-type="bibr" rid="ref56">56</xref>]. The output quality and stability of AI chatbots are positively associated with their ability to structure clinical knowledge and are influenced by task complexity, further supporting the emphasis on transparency, usability, and validity in explainable AI systems for clinical decision support [<xref ref-type="bibr" rid="ref57">57</xref>].</p></sec><sec id="s4-2"><title>Limitations</title><p>This study has several limitations. First, we evaluated AI chatbots performance on moderately severe and well-documented ROP cases (excluding stage 4 and above) rather than the full clinical spectrum encountered in routine ROP screening scenarios, failing to reflect real-world data quality challenges and limiting the clinical applicability of our findings. Furthermore, the single-center convenience sampling design and potential selection bias restrict geographic and demographic generalizability. Notably, although some findings from the Zone I subgroup analysis reached statistical significance, these results should be interpreted with caution due to the small sample size, and validation in larger cohorts is still needed. Second, given the exploratory nature of this study, formal adjustments for multiple comparisons (eg, Bonferroni correction) were not applied to avoid an inflated risk of Type II error (false-negative conclusions). While this represents a potential limitation, the robust effect sizes observed for the primary endpoints largely mitigated this concern. Third, the lack of cross-platform validation across different software and hardware environments limits the generalizability of the AI chatbots and their reliability in diverse clinical settings. Fourth, the AI chatbots have different training data cutoffs, resulting in asynchronous knowledge updates that may be inconsistent with contemporary clinical practice standards, with Gemini 2.5 Pro trained up to January 2025 and ChatGPT o4-mini up to June 2024. This discrepancy in training data cutoff dates between the two AI chatbots may also contribute to the observed differences in their diagnostic performance. Fifth, there is still a potential model hierarchy mismatch between Gemini 2.5 Pro (flagship high-computation model) and ChatGPT o4-mini (efficiency-focused model), despite our use of the enhanced reasoning mode of ChatGPT o4-mini. Sixth, our prompt instructions forced the 2 AI chatbots to omit analytical processes and directly output results, which may have artificially lowered their performance. The prompts did not explicitly request eye-specific separate treatment suggestions, though the models spontaneously generated treatment suggestions for each eye separately, which may affect the accuracy of our analysis results. Furthermore, in the treatment suggestion task, we manually input gold-standard diagnoses rather than using the chatbots&#x2019; own outputs, and all diagnostic and therapeutic queries were completed within a single continuous chat session, which precluded evaluation of their end-to-end clinical management performance. All interactions were continuous, so the treatment suggestions generated by chatbots were not produced within an isolated conversational context. Moreover, all interactions were conducted in Chinese, and the AI chatbot performance may vary across prompting languages. Finally, the absence of standardized tools to quantify &#x201C;hallucinations in treatment suggestion&#x201D; prevents objective assessment of potential risks in chatbot outputs, weakening their credibility as clinical decision support tools.</p></sec><sec id="s4-3"><title>Future Directions</title><p>While our results, validated against specialist consensus as the gold standard, appeared promising, rigorous external validation&#x2014;including multicenter studies, prospective trials, and cross-platform verification&#x2014;is essential before clinical implementation to confirm generalizability and ensure patient safety. Future studies can validate the end-to-end clinical management performance of AI chatbots based on their own independent diagnoses, rather than manual input of gold-standard diagnoses. In addition, further balanced cross-model comparisons are also needed to explore how model size, parameter count, and architecture shape the performance of AI chatbots in multimodal ROP tasks.</p></sec><sec id="s4-4"><title>Conclusion</title><p>In this study of ROP, ChatGPT o4-mini appears to be more promising for evidence-based treatment suggestions based on the gold standard diagnoses, while Gemini 2.5 Pro&#x2019;s visual acuity highlights its potential for targeted ROP diagnostic screening, particularly in identifying plus disease. As these AI chatbots continue to evolve, further validation in larger and more diverse cohorts is warranted to establish their clinical utility and generalizability.</p></sec></sec></body><back><ack><p>We used Gemini 2.5 Pro (Google) [<xref ref-type="bibr" rid="ref33">33</xref>] and ChatGPT o4-mini (OpenAI) [<xref ref-type="bibr" rid="ref49">49</xref>] to generate response interfaces for infants with retinopathy of prematurity (ROP). These responses were subsequently reviewed and revised by the research team. <xref ref-type="fig" rid="figure2">Figures 2 and 3</xref> were generated using Gemini 2.5 Pro (Google) [<xref ref-type="bibr" rid="ref33">33</xref>] and ChatGPT o4-mini (OpenAI) [<xref ref-type="bibr" rid="ref49">49</xref>].</p></ack><notes><sec><title>Funding</title><p>This work was supported by Shenzhen Medical Research Fund (C2301005, C2501034, A2403020), National Natural Science Foundation of China (82301269, 82271103, 82401315, 82401272, 82301226), Shenzhen Science and Technology R&#x0026;D Fund Program (JCYJ20240813152703005, JCYJ20250604184009011), Guangdong Basic and Applied Basic Research Foundation (2026A1515012558), Sanming Project of Medicine in Shenzhen (SZSM202311018).</p></sec><sec><title>Data Availability</title><p>Core data are provided in the manuscript. The raw AI outputs generated during this study, including diagnostic and treatment suggestions from both models, are available from the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: SP, XZ, ZW, and GZ.</p><p>Methodology: SP, XZ, ZW, and GZ.</p><p>Investigation: DY and ND.</p><p>Resources: DY and ND.</p><p>Validation: DY and ND.</p><p>Software: SP and XZ.</p><p>Formal analysis: SP and XZ.</p><p>Data curation: SP and XZ.</p><p>Visualization: SP and XZ.</p><p>Writing &#x2013; original draft: SP and XZ.</p><p>Writing &#x2013; review and editing: SP, XZ, ZW, and GZ.</p><p>Funding acquisition: KC, ZY, WY, WW, and WC.</p><p>Project administration: KC, ZY, WY, WW, and WC.</p><p>Supervision: KC, ZY, WY, WW, and WC.</p><p>GZ, WC, WY contributed equally as the corresponding authors.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BW</term><def><p>birth weight</p></def></def-item><def-item><term id="abb2">GA</term><def><p>gestational age</p></def></def-item><def-item><term id="abb3">GEE</term><def><p>generalized estimating equation</p></def></def-item><def-item><term id="abb4">GQS</term><def><p>Global Quality Score</p></def></def-item><def-item><term id="abb5">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb6">ICROP3</term><def><p>the third edition of the International Classification of ROP</p></def></def-item><def-item><term id="abb7">IRB</term><def><p>Institutional Review Board</p></def></def-item><def-item><term id="abb8">OR</term><def><p>odds ratio</p></def></def-item><def-item><term id="abb9">ROP</term><def><p>retinopathy of prematurity</p></def></def-item><def-item><term id="abb10">SZEH </term><def><p>Shenzhen Eye Hospital</p></def></def-item><def-item><term id="abb11">VEGF</term><def><p>vascular endothelial growth factor</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gilbert</surname><given-names>C</given-names> </name></person-group><article-title>Retinopathy of prematurity: a global perspective of the epidemics, population of babies at risk and implications for control</article-title><source>Early Hum Dev</source><year>2008</year><month>02</month><volume>84</volume><issue>2</issue><fpage>77</fpage><lpage>82</lpage><pub-id pub-id-type="doi">10.1016/j.earlhumdev.2007.11.009</pub-id><pub-id pub-id-type="medline">18234457</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Blencowe</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lawn</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Vazquez</surname><given-names>T</given-names> </name><name name-style="western"><surname>Fielder</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gilbert</surname><given-names>C</given-names> </name></person-group><article-title>Preterm-associated visual impairment and estimates of retinopathy of prematurity at regional and global levels for 2010</article-title><source>Pediatr Res</source><year>2013</year><month>12</month><volume>74 Suppl 1</volume><issue>Suppl 1</issue><fpage>35</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1038/pr.2013.205</pub-id><pub-id pub-id-type="medline">24366462</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garc&#x00ED;a</surname><given-names>H</given-names> </name><name name-style="western"><surname>Villasis-Keever</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Zavala-Vargas</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bravo-Ortiz</surname><given-names>JC</given-names> </name><name name-style="western"><surname>P&#x00E9;rez-M&#x00E9;ndez</surname><given-names>A</given-names> </name><name name-style="western"><surname>Escamilla-N&#x00FA;&#x00F1;ez</surname><given-names>A</given-names> </name></person-group><article-title>Global prevalence and severity of retinopathy of prematurity over the last four decades (1985-2021): a systematic review and meta-analysis</article-title><source>Arch Med Res</source><year>2024</year><month>02</month><volume>55</volume><issue>2</issue><fpage>102967</fpage><pub-id pub-id-type="doi">10.1016/j.arcmed.2024.102967</pub-id><pub-id pub-id-type="medline">38364488</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wood</surname><given-names>EH</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>EY</given-names> </name><name name-style="western"><surname>Beck</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hadfield</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Quinn</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Harper</surname><given-names>CA</given-names>  <suffix>3rd</suffix></name></person-group><article-title>80 Years of vision: preventing blindness from retinopathy of prematurity</article-title><source>J Perinatol</source><year>2021</year><month>06</month><volume>41</volume><issue>6</issue><fpage>1216</fpage><lpage>1224</lpage><pub-id pub-id-type="doi">10.1038/s41372-021-01015-8</pub-id><pub-id pub-id-type="medline">33674712</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Quinn</surname><given-names>GE</given-names> </name><name name-style="western"><surname>Ying</surname><given-names>G shuang</given-names> </name><name name-style="western"><surname>Daniel</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Validity of a telemedicine system for the evaluation of acute-phase retinopathy of prematurity</article-title><source>JAMA Ophthalmol</source><year>2014</year><month>10</month><volume>132</volume><issue>10</issue><fpage>1178</fpage><lpage>1184</lpage><pub-id pub-id-type="doi">10.1001/jamaophthalmol.2014.1604</pub-id><pub-id pub-id-type="medline">24970095</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vinekar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jayadev</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mangalesh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shetty</surname><given-names>B</given-names> </name><name name-style="western"><surname>Vidyasagar</surname><given-names>D</given-names> </name></person-group><article-title>Role of tele-medicine in retinopathy of prematurity screening in rural outreach centers in India - a report of 20,214 imaging sessions in the KIDROP program</article-title><source>Semin Fetal Neonatal Med</source><year>2015</year><month>10</month><volume>20</volume><issue>5</issue><fpage>335</fpage><lpage>345</lpage><pub-id pub-id-type="doi">10.1016/j.siny.2015.05.002</pub-id><pub-id pub-id-type="medline">26092301</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hennein</surname><given-names>L</given-names> </name><name name-style="western"><surname>Jastrzembski</surname><given-names>B</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>AS</given-names> </name></person-group><article-title>Use of telemedicine in pediatric ophthalmology in the underserved population</article-title><source>Semin Ophthalmol</source><year>2023</year><month>02</month><volume>38</volume><issue>2</issue><fpage>116</fpage><lpage>123</lpage><pub-id pub-id-type="doi">10.1080/08820538.2022.2152703</pub-id><pub-id pub-id-type="medline">36529958</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Takeda</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kaneko</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sugimoto</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yamashita</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sasaki</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mitsui</surname><given-names>T</given-names> </name></person-group><article-title>Prediction models for retinopathy of prematurity using nonimaging machine learning approaches: a regional multicenter study</article-title><source>Ophthalmol Sci</source><year>2025</year><volume>5</volume><issue>4</issue><fpage>100715</fpage><pub-id pub-id-type="doi">10.1016/j.xops.2025.100715</pub-id><pub-id pub-id-type="medline">40182984</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barrero-Castillero</surname><given-names>A</given-names> </name><name name-style="western"><surname>Corwin</surname><given-names>BK</given-names> </name><name name-style="western"><surname>VanderVeen</surname><given-names>DK</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>JC</given-names> </name></person-group><article-title>Workforce shortage for retinopathy of prematurity care and emerging role of telehealth and artificial intelligence</article-title><source>Pediatr Clin North Am</source><year>2020</year><month>08</month><volume>67</volume><issue>4</issue><fpage>725</fpage><lpage>733</lpage><pub-id pub-id-type="doi">10.1016/j.pcl.2020.04.012</pub-id><pub-id pub-id-type="medline">32650869</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fouzdar Jain</surname><given-names>S</given-names> </name><name name-style="western"><surname>Song</surname><given-names>HH</given-names> </name><name name-style="western"><surname>Al-Holou</surname><given-names>SN</given-names> </name><name name-style="western"><surname>Morgan</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Suh</surname><given-names>DW</given-names> </name></person-group><article-title>Retinopathy of prematurity: preferred practice patterns among pediatric ophthalmologists</article-title><source>Clin Ophthalmol</source><year>2018</year><volume>12</volume><fpage>1003</fpage><lpage>1009</lpage><pub-id pub-id-type="doi">10.2147/OPTH.S161504</pub-id><pub-id pub-id-type="medline">29881255</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chiang</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Quinn</surname><given-names>GE</given-names> </name><name name-style="western"><surname>Fielder</surname><given-names>AR</given-names> </name><etal/></person-group><article-title>International Classification of Retinopathy of Prematurity, Third Edition</article-title><source>Ophthalmology</source><year>2021</year><month>10</month><volume>128</volume><issue>10</issue><fpage>e51</fpage><lpage>e68</lpage><pub-id pub-id-type="doi">10.1016/j.ophtha.2021.05.031</pub-id><pub-id pub-id-type="medline">34247850</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fierson</surname><given-names>WM</given-names> </name><collab>American Academy of Pediatrics Section on Ophthalmology</collab><collab>American Academy of Ophthalmology</collab><collab>American Association for Pediatric Ophthalmology and Strabismus</collab><collab>American Association of Certified Orthoptists</collab></person-group><article-title>Screening examination of premature infants for retinopathy of prematurity</article-title><source>Pediatrics</source><year>2018</year><month>12</month><volume>142</volume><issue>6</issue><fpage>e20183061</fpage><pub-id pub-id-type="doi">10.1542/peds.2018-3061</pub-id><pub-id pub-id-type="medline">30478242</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abidin</surname><given-names>ZU</given-names> </name><name name-style="western"><surname>Naqvi</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Jeong</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SW</given-names> </name></person-group><article-title>Optimizing optic cup and optic disc delineation: Introducing the efficient feature preservation segmentation network</article-title><source>Eng Appl Artif Intell</source><year>2025</year><month>03</month><volume>144</volume><fpage>110038</fpage><pub-id pub-id-type="doi">10.1016/j.engappai.2025.110038</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Evaluating large language models in ophthalmology: systematic review</article-title><source>J Med Internet Res</source><year>2025</year><month>10</month><day>27</day><volume>27</volume><fpage>e76947</fpage><pub-id pub-id-type="doi">10.2196/76947</pub-id><pub-id pub-id-type="medline">41144954</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Grzybowski</surname><given-names>A</given-names> </name></person-group><article-title>Multimodal artificial intelligence in ophthalmology: applications, challenges, and future directions</article-title><source>Surv Ophthalmol</source><year>2026</year><volume>71</volume><issue>1</issue><fpage>158</fpage><lpage>167</lpage><pub-id pub-id-type="doi">10.1016/j.survophthal.2025.07.003</pub-id><pub-id pub-id-type="medline">40683606</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Kan</surname><given-names>H</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Geng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Nie</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>M</given-names> </name></person-group><article-title>MED-ChatGPT CoPilot: a ChatGPT medical assistant for case mining and adjunctive therapy</article-title><source>Front Med (Lausanne)</source><year>2024</year><volume>11</volume><fpage>1460553</fpage><pub-id pub-id-type="doi">10.3389/fmed.2024.1460553</pub-id><pub-id pub-id-type="medline">39478827</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Choudhary</surname><given-names>HK</given-names> </name><name name-style="western"><surname>Hardin</surname><given-names>WM</given-names> </name><name name-style="western"><surname>Prakash</surname><given-names>N</given-names> </name></person-group><article-title>Comparative analysis of ChatGPT-4o and Gemini Advanced performance on diagnostic radiology in-training exams</article-title><source>Cureus</source><year>2025</year><month>03</month><volume>17</volume><issue>3</issue><fpage>e80874</fpage><pub-id pub-id-type="doi">10.7759/cureus.80874</pub-id><pub-id pub-id-type="medline">40255788</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sandmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hegselmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fujarski</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Benchmark evaluation of DeepSeek large language models in clinical decision-making</article-title><source>Nat Med</source><year>2025</year><month>08</month><volume>31</volume><issue>8</issue><fpage>2546</fpage><lpage>2549</lpage><pub-id pub-id-type="doi">10.1038/s41591-025-03727-2</pub-id><pub-id pub-id-type="medline">40267970</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>David</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zloto</surname><given-names>O</given-names> </name><name name-style="western"><surname>Katz</surname><given-names>G</given-names> </name><etal/></person-group><article-title>The use of artificial intelligence based chat bots in ophthalmology triage</article-title><source>Eye (Lond)</source><year>2025</year><month>03</month><volume>39</volume><issue>4</issue><fpage>785</fpage><lpage>789</lpage><pub-id pub-id-type="doi">10.1038/s41433-024-03488-1</pub-id><pub-id pub-id-type="medline">39592814</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tomita</surname><given-names>K</given-names> </name><name name-style="western"><surname>Nishida</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kitaguchi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kitazawa</surname><given-names>K</given-names> </name><name name-style="western"><surname>Miyake</surname><given-names>M</given-names> </name></person-group><article-title>Image recognition performance of GPT-4V(ision) and GPT-4o in ophthalmology: use of images in clinical questions</article-title><source>Clin Ophthalmol</source><year>2025</year><volume>19</volume><fpage>1557</fpage><lpage>1564</lpage><pub-id pub-id-type="doi">10.2147/OPTH.S494480</pub-id><pub-id pub-id-type="medline">40357454</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ma</surname><given-names>R</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Yao</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Multimodal machine learning enables AI chatbot to diagnose ophthalmic diseases and provide high-quality medical responses</article-title><source>NPJ Digit Med</source><year>2025</year><month>01</month><day>27</day><volume>8</volume><issue>1</issue><fpage>64</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01461-0</pub-id><pub-id pub-id-type="medline">39870855</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Carl&#x00E0;</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Gambini</surname><given-names>G</given-names> </name><name name-style="western"><surname>Baldascino</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Exploring AI-chatbots&#x2019; capability to suggest surgical planning in ophthalmology: ChatGPT versus Google Gemini analysis of retinal detachment cases</article-title><source>Br J Ophthalmol</source><year>2024</year><month>09</month><day>20</day><volume>108</volume><issue>10</issue><fpage>1457</fpage><lpage>1469</lpage><pub-id pub-id-type="doi">10.1136/bjo-2023-325143</pub-id><pub-id pub-id-type="medline">38448201</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Duan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Exploring the capabilities of three artificial intelligence chatbots in diagnosis and decision-making of age-related macular degeneration</article-title><source>Br J Ophthalmol</source><year>2026</year><month>06</month><day>10</day><fpage>bjo-2025-328860</fpage><pub-id pub-id-type="doi">10.1136/bjo-2025-328860</pub-id><pub-id pub-id-type="medline">42270292</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gupta</surname><given-names>A</given-names> </name><name name-style="western"><surname>Al-Kazwini</surname><given-names>H</given-names> </name></person-group><article-title>Evaluating ChatGPT&#x2019;s diagnostic accuracy in detecting fundus images</article-title><source>Cureus</source><year>2024</year><volume>16</volume><fpage>e73660</fpage><pub-id pub-id-type="doi">10.7759/cureus.73660</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Eyecare-cloud: an innovative electronic medical record cloud platform for pediatric research and clinical care</article-title><source>EPMA J</source><year>2024</year><month>09</month><volume>15</volume><issue>3</issue><fpage>501</fpage><lpage>510</lpage><pub-id pub-id-type="doi">10.1007/s13167-024-00372-6</pub-id><pub-id pub-id-type="medline">39239111</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>K</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>He</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>D</given-names> </name></person-group><article-title>DeepSeek-R1 outperforms Gemini 2.0 Pro, OpenAI o1, and o3-mini in bilingual complex ophthalmology reasoning</article-title><source>Adv Ophthalmol Pract Res</source><year>2025</year><volume>5</volume><issue>3</issue><fpage>189</fpage><lpage>195</lpage><pub-id pub-id-type="doi">10.1016/j.aopr.2025.05.001</pub-id><pub-id pub-id-type="medline">40678192</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bellanda</surname><given-names>VCF</given-names> </name><name name-style="western"><surname>Santos</surname><given-names>MLD</given-names> </name><name name-style="western"><surname>Ferraz</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Jorge</surname><given-names>R</given-names> </name><name name-style="western"><surname>Melo</surname><given-names>GB</given-names> </name></person-group><article-title>Applications of ChatGPT in the diagnosis, management, education, and research of retinal diseases: a scoping review</article-title><source>Int J Retina Vitreous</source><year>2024</year><month>10</month><day>17</day><volume>10</volume><issue>1</issue><fpage>79</fpage><pub-id pub-id-type="doi">10.1186/s40942-024-00595-9</pub-id><pub-id pub-id-type="medline">39420407</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li&#x00E9;vin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Hother</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Motzfeldt</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Winther</surname><given-names>O</given-names> </name></person-group><article-title>Can large language models reason about medical questions?</article-title><source>Patterns (NY)</source><year>2024</year><month>03</month><day>8</day><volume>5</volume><issue>3</issue><fpage>100943</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2024.100943</pub-id><pub-id pub-id-type="medline">38487804</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Joseph</surname><given-names>A</given-names> </name><name name-style="western"><surname>Joseph</surname><given-names>K</given-names> </name><name name-style="western"><surname>Joseph</surname><given-names>A</given-names> </name></person-group><article-title>A pilot evaluation of the diagnostic accuracy of ChatGPT-3.5 for multiple sclerosis from case reports</article-title><source>Transl Neurosci</source><year>2024</year><month>01</month><day>1</day><volume>15</volume><issue>1</issue><fpage>20220361</fpage><pub-id pub-id-type="doi">10.1515/tnsci-2022-0361</pub-id><pub-id pub-id-type="medline">39726894</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>CH</given-names> </name><name name-style="western"><surname>Hsieh</surname><given-names>KY</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>HY</given-names> </name></person-group><article-title>Comparing vision-capable models, GPT-4 and Gemini, with GPT-3.5 on Taiwan&#x2019;s pulmonologist exam</article-title><source>Cureus</source><year>2024</year><month>08</month><volume>16</volume><issue>8</issue><fpage>e67641</fpage><pub-id pub-id-type="doi">10.7759/cureus.67641</pub-id><pub-id pub-id-type="medline">39185287</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Silhadi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Nassrallah</surname><given-names>WB</given-names> </name><name name-style="western"><surname>Mikhail</surname><given-names>D</given-names> </name><name name-style="western"><surname>Milad</surname><given-names>D</given-names> </name><name name-style="western"><surname>Harissi-Dagher</surname><given-names>M</given-names> </name></person-group><article-title>Assessing the performance of Microsoft Copilot, GPT-4 and Google Gemini in ophthalmology</article-title><source>Can J Ophthalmol</source><year>2025</year><month>08</month><volume>60</volume><issue>4</issue><fpage>e507</fpage><lpage>e514</lpage><pub-id pub-id-type="doi">10.1016/j.jcjo.2025.01.001</pub-id><pub-id pub-id-type="medline">39863285</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Antaki</surname><given-names>F</given-names> </name><name name-style="western"><surname>Chopra</surname><given-names>R</given-names> </name><name name-style="western"><surname>Keane</surname><given-names>PA</given-names> </name></person-group><article-title>Vision-language models for feature detection of macular diseases on optical coherence tomography</article-title><source>JAMA Ophthalmol</source><year>2024</year><month>06</month><day>1</day><volume>142</volume><issue>6</issue><fpage>573</fpage><lpage>576</lpage><pub-id pub-id-type="doi">10.1001/jamaophthalmol.2024.1165</pub-id><pub-id pub-id-type="medline">38696177</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="web"><article-title>Gemini 25 Pro</article-title><source>Google Cloud</source><access-date>2026-09-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://cloud.google.com/vertex-ai/generative-ai/docs/models/gemini/2-5-pro">https://cloud.google.com/vertex-ai/generative-ai/docs/models/gemini/2-5-pro</ext-link></comment></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chia</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Antaki</surname><given-names>F</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Turner</surname><given-names>AW</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>AY</given-names> </name><name name-style="western"><surname>Keane</surname><given-names>PA</given-names> </name></person-group><article-title>Foundation models in ophthalmology</article-title><source>Br J Ophthalmol</source><year>2024</year><month>09</month><day>20</day><volume>108</volume><issue>10</issue><fpage>1341</fpage><lpage>1348</lpage><pub-id pub-id-type="doi">10.1136/bjo-2024-325459</pub-id><pub-id pub-id-type="medline">38834291</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name></person-group><article-title>Dissecting HealthBench: disease spectrum, clinical diversity, and data insights from multi-turn clinical AI evaluation benchmark</article-title><source>J Med Syst</source><year>2025</year><month>07</month><day>28</day><volume>49</volume><issue>1</issue><fpage>100</fpage><pub-id pub-id-type="doi">10.1007/s10916-025-02232-w</pub-id><pub-id pub-id-type="medline">40719790</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="web"><article-title>Introducing HealthBench</article-title><source>OpenAI</source><year>2025</year><access-date>2026-09-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openai.com/index/healthbench">https://openai.com/index/healthbench</ext-link></comment></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="web"><article-title>OpenAI o3 and o4-mini System Card</article-title><source>OpenAI</source><year>2025</year><access-date>2026-09-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openai.com/index/o3-o4-mini-system-card">https://openai.com/index/o3-o4-mini-system-card</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="web"><article-title>O4-mini: fast, cost-efficient reasoning model</article-title><source>OpenAI</source><year>2025</year><access-date>2026-09-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://developers.openai.com/api/docs/models/o4-mini">https://developers.openai.com/api/docs/models/o4-mini</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bernard</surname><given-names>A</given-names> </name><name name-style="western"><surname>Langille</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hughes</surname><given-names>S</given-names> </name><name name-style="western"><surname>Rose</surname><given-names>C</given-names> </name><name name-style="western"><surname>Leddin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Veldhuyzen van Zanten</surname><given-names>S</given-names> </name></person-group><article-title>A systematic review of patient inflammatory bowel disease information resources on the world wide web</article-title><source>Am J Gastroenterol</source><year>2007</year><month>09</month><volume>102</volume><issue>9</issue><fpage>2070</fpage><lpage>2077</lpage><pub-id pub-id-type="doi">10.1111/j.1572-0241.2007.01325.x</pub-id><pub-id pub-id-type="medline">17511753</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reyhan</surname><given-names>AH</given-names> </name><name name-style="western"><surname>Mutaf</surname><given-names>&#x00C7;</given-names> </name><name name-style="western"><surname>Uzun</surname><given-names>&#x0130;</given-names> </name><name name-style="western"><surname>Y&#x00FC;ksekyayla</surname><given-names>F</given-names> </name></person-group><article-title>A performance evaluation of large language models in keratoconus: a comparative study of ChatGPT-3.5, ChatGPT-4.0, Gemini, Copilot, Chatsonic, and Perplexity</article-title><source>J Clin Med</source><year>2024</year><month>10</month><day>30</day><volume>13</volume><issue>21</issue><fpage>6512</fpage><pub-id pub-id-type="doi">10.3390/jcm13216512</pub-id><pub-id pub-id-type="medline">39518652</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rao</surname><given-names>M</given-names> </name><name name-style="western"><surname>Xiujun</surname><given-names>T</given-names> </name><name name-style="western"><surname>Haoyu</surname><given-names>W</given-names> </name></person-group><article-title>Evaluating GPT-4 responses on scars or keloids for patient education: large language model evaluation study</article-title><source>JMIR Med Inform</source><year>2026</year><month>02</month><day>27</day><volume>14</volume><fpage>e78838</fpage><pub-id pub-id-type="doi">10.2196/78838</pub-id><pub-id pub-id-type="medline">41773665</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maywood</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Parikh</surname><given-names>R</given-names> </name><name name-style="western"><surname>Deobhakta</surname><given-names>A</given-names> </name><name name-style="western"><surname>Begaj</surname><given-names>T</given-names> </name></person-group><article-title>Performance assessment of an artificial intelligence chatbot in clinical vitreoretinal scenarios</article-title><source>Retina</source><year>2024</year><month>06</month><day>1</day><volume>44</volume><issue>6</issue><fpage>954</fpage><lpage>964</lpage><pub-id pub-id-type="doi">10.1097/IAE.0000000000004053</pub-id><pub-id pub-id-type="medline">38271674</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abbas</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Seol</surname><given-names>H</given-names> </name><name name-style="western"><surname>Abbas</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SW</given-names> </name></person-group><article-title>Exploring the role of artificial intelligence in smart healthcare: a capability and function-oriented review</article-title><source>Healthcare (Basel)</source><year>2025</year><month>07</month><day>8</day><volume>13</volume><issue>14</issue><fpage>1642</fpage><pub-id pub-id-type="doi">10.3390/healthcare13141642</pub-id><pub-id pub-id-type="medline">40724669</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Campbell</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Chiang</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>JS</given-names> </name><etal/></person-group><article-title>Artificial intelligence for retinopathy of prematurity: validation of a vascular severity scale against international expert diagnosis</article-title><source>Ophthalmology</source><year>2022</year><month>07</month><volume>129</volume><issue>7</issue><fpage>e69</fpage><lpage>e76</lpage><pub-id pub-id-type="doi">10.1016/j.ophtha.2022.02.008</pub-id><pub-id pub-id-type="medline">35157950</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hartsock</surname><given-names>I</given-names> </name><name name-style="western"><surname>Rasool</surname><given-names>G</given-names> </name></person-group><article-title>Vision-language models for medical report generation and visual question answering: a review</article-title><source>Front Artif Intell</source><year>2024</year><volume>7</volume><fpage>1430984</fpage><pub-id pub-id-type="doi">10.3389/frai.2024.1430984</pub-id><pub-id pub-id-type="medline">39628839</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Campbell</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Ataer-Cansizoglu</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bolon-Canedo</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Expert diagnosis of plus disease in retinopathy of prematurity from computer-based image analysis</article-title><source>JAMA Ophthalmol</source><year>2016</year><month>06</month><day>1</day><volume>134</volume><issue>6</issue><fpage>651</fpage><lpage>657</lpage><pub-id pub-id-type="doi">10.1001/jamaophthalmol.2016.0611</pub-id><pub-id pub-id-type="medline">27077667</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhuge</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>MoE-Adapters++: toward more efficient continual learning of vision-language models via dynamic mixture-of-experts adapters</article-title><source>IEEE Trans Pattern Anal Mach Intell</source><year>2025</year><month>12</month><volume>47</volume><issue>12</issue><fpage>11912</fpage><lpage>11928</lpage><pub-id pub-id-type="doi">10.1109/TPAMI.2025.3597942</pub-id><pub-id pub-id-type="medline">40788794</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><name name-style="western"><surname>He</surname><given-names>X</given-names> </name><etal/></person-group><article-title>VALOR: Vision-audio-language omni-perception pretraining model and dataset</article-title><source>IEEE Trans Pattern Anal Mach Intell</source><year>2025</year><volume>47</volume><issue>2</issue><fpage>708</fpage><lpage>724</lpage><pub-id pub-id-type="doi">10.1109/TPAMI.2024.3479776</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="web"><article-title>Introducing openai o3 and o4-mini</article-title><source>OpenAI</source><access-date>2026-09-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openai.com/index/introducing-o3-and-o4-mini">https://openai.com/index/introducing-o3-and-o4-mini</ext-link></comment></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koga</surname><given-names>S</given-names> </name><name name-style="western"><surname>Du</surname><given-names>W</given-names> </name></person-group><article-title>Challenges of integrating chatbot use in ophthalmology diagnostics</article-title><source>JAMA Ophthalmol</source><year>2024</year><month>09</month><day>1</day><volume>142</volume><issue>9</issue><fpage>883</fpage><lpage>884</lpage><pub-id pub-id-type="doi">10.1001/jamaophthalmol.2024.2303</pub-id><pub-id pub-id-type="medline">38958958</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cai</surname><given-names>LZ</given-names> </name><name name-style="western"><surname>Shaheen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Performance of generative large language models on ophthalmology board-style questions</article-title><source>Am J Ophthalmol</source><year>2023</year><month>10</month><volume>254</volume><fpage>141</fpage><lpage>149</lpage><pub-id pub-id-type="doi">10.1016/j.ajo.2023.05.024</pub-id><pub-id pub-id-type="medline">37339728</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Song</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>VisionUnite: a vision-language foundation model for ophthalmology enhanced with clinical knowledge</article-title><source>IEEE Trans Pattern Anal Mach Intell</source><year>2025</year><volume>47</volume><issue>12</issue><fpage>11848</fpage><lpage>11862</lpage><pub-id pub-id-type="doi">10.1109/TPAMI.2025.3598734</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gottweis</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="medline">39779926</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kung</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Cheatham</surname><given-names>M</given-names> </name><name name-style="western"><surname>Medenilla</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models</article-title><source>PLOS Digit Health</source><year>2023</year><month>02</month><volume>2</volume><issue>2</issue><fpage>e0000198</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000198</pub-id><pub-id pub-id-type="medline">36812645</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Transforming free-text radiology reports into structured reports using ChatGPT: a study on thyroid ultrasonography</article-title><source>Eur J Radiol</source><year>2024</year><month>06</month><volume>175</volume><fpage>111458</fpage><pub-id pub-id-type="doi">10.1016/j.ejrad.2024.111458</pub-id><pub-id pub-id-type="medline">38613868</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Comparative performance of large language models for patient-initiated ophthalmology consultations</article-title><source>Front Public Health</source><year>2025</year><volume>13</volume><fpage>1673045</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2025.1673045</pub-id><pub-id pub-id-type="medline">41059182</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abbas</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Jeong</surname><given-names>W</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SW</given-names> </name></person-group><article-title>Explainable AI in clinical decision support systems: a meta-analysis of methods, applications, and usability challenges</article-title><source>Healthcare (Basel)</source><year>2025</year><month>08</month><day>29</day><volume>13</volume><issue>17</issue><fpage>2154</fpage><pub-id pub-id-type="doi">10.3390/healthcare13172154</pub-id><pub-id pub-id-type="medline">40941506</pub-id></nlm-citation></ref></ref-list></back></article>