<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e97802</article-id><article-id pub-id-type="doi">10.2196/97802</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Performance of Large Language Models for Oncology Nursing Decision Support: Cross-Sectional Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Zhou</surname><given-names>Qiongyu</given-names></name><degrees>BSN</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jia</surname><given-names>Yan</given-names></name><degrees>MSN</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hu</surname><given-names>Haiqin</given-names></name><degrees>MSN</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Huang</surname><given-names>Danjiao</given-names></name><degrees>BSN</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Xuefeng</given-names></name><degrees>BSN</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Xia</surname><given-names>Yirong</given-names></name><degrees>BSN</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Wu</surname><given-names>Wanying</given-names></name><degrees>MSN</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>School of Nursing, Zhejiang Chinese Medical University</institution><addr-line>Hangzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><aff id="aff2"><institution>Hangzhou Institute of Medicine, Chinese Academy of Sciences, Zhejiang Cancer Hospital</institution><addr-line>Hangzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Rosa</surname><given-names>Delaney La</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Koike</surname><given-names>Takeshi</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Wanying Wu, MSN, Hangzhou Institute of Medicine, Chinese Academy of Sciences, Zhejiang Cancer Hospital, Hangzhou, Zhejiang, 310022, China, 86 13857137426; <email>764286275@qq.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>24</day><month>7</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e97802</elocation-id><history><date date-type="received"><day>10</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>02</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>02</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Qiongyu Zhou, Yan Jia, Haiqin Hu, Danjiao Huang, Xuefeng Chen, Yirong Xia, Wanying Wu. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 24.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e97802"/><abstract><sec><title>Background</title><p>Large language models (LLMs) are increasingly used in health care, with emerging applications in clinical decision support and nursing education. However, evidence on their performance in nursing contexts, particularly in oncology nursing, remains limited. Given the complexity and high-risk nature of oncology care, it is important to evaluate the performance and clinical relevance of LLM-generated responses in oncology nursing contexts.</p></sec><sec><title>Objective</title><p>This study aimed to compare the performance of LLMs in oncology nursing decision support tasks using standardized examination questions and case-based clinical scenarios and explore LLMs&#x2019; potential applicability and current limitations in oncology nursing practice.</p></sec><sec sec-type="methods"><title>Methods</title><p>A total of 33 case-based questions derived from 10 oncology nursing clinical scenarios in a nationally used training manual, along with standardized examination-oriented questions from a commercially published preparation book for the Chinese Nursing (Intermediate) Qualification Examination, were used to evaluate the performance of 5 LLMs (DeepSeek, Qwen, Spark-Desk, WiseDiag, and ChatGPT). All models generated responses using a standardized prompt. Two oncology nurses with more than 5 years of clinical experience independently rated the case-based responses using 3 evaluation dimensions: correctness, clarity, and conciseness. Interrater reliability was assessed using the quadratic weighted Cohen &#x03BA;, intraclass correlation coefficient, and Spearman rank correlation coefficient. Differences among models were analyzed using the Kruskal-Wallis test with the Dunn post hoc test. In addition, examination performance was evaluated based on total score, accuracy rate, and completion efficiency.</p></sec><sec sec-type="results"><title>Results</title><p>Interrater reliability analyses indicated moderate agreement between evaluators. The median correctness, clarity, and conciseness scores were as follows: 11.50 (IQR 10.50-12.00) for DeepSeek, 11.00 (IQR 10.50-12.00) for Qwen, 10.50 (IQR 9.50-11.50) for Spark-Desk, 10.00 (IQR 9.50-11.50) for WiseDiag, and 10.00 (IQR 9.00-11.50) for ChatGPT. The Kruskal-Wallis test indicated statistically significant differences among models (<italic>H</italic>=11.416; <italic>P</italic>&#x003C;.05), with post hoc analysis showing a significant difference only between DeepSeek and ChatGPT (<italic>P</italic>&#x003C;.05). In examination-based tasks, all models achieved passing performance, with accuracy rates ranging from 77% (77/100) to 93% (93/100). In terms of response completion, DeepSeek and ChatGPT completed all tasks in a single interaction, whereas other models required multiple interactions due to output interruptions.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>LLMs showed relatively strong performance on structured knowledge and examination-based tasks but remained limited in complex oncology nursing scenarios requiring individualized assessment and dynamic clinical judgment. Their potential use may be most relevant to information retrieval, knowledge organization, and patient education. Because the correctness, clarity, and conciseness rubric showed only moderate interrater reliability, the case-based comparisons should be interpreted as preliminary signals rather than definitive evidence of between-model differences. LLM outputs should therefore be used as supportive information and interpreted alongside professional clinical judgment.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>artificial intelligence</kwd><kwd>AI</kwd><kwd>natural language processing</kwd><kwd>nursing</kwd><kwd>oncology nursing</kwd><kwd>clinical decision support systems</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Oncology nursing is characterized by complex symptom management, rapidly evolving patient conditions, and high-stakes clinical decision-making [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Nurses are required to continuously integrate multidimensional information, including treatment regimens, adverse effects, and patient-specific factors, often under conditions of uncertainty and time pressure. As cancer care becomes increasingly intensive and individualized, the cognitive demands placed on oncology nurses continue to grow, highlighting the need for efficient and effective clinical decision support tools [<xref ref-type="bibr" rid="ref3">3</xref>].</p><p>In this context, AI has rapidly advanced and is increasingly integrated into health care practice [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. Among these developments, large language models (LLMs) have attracted particular attention due to their capacity to process natural language, synthesize dispersed information, and generate contextually relevant responses. Unlike traditional retrieval-based systems, LLMs enable interactive, dialogue-based support, allowing clinicians to access synthesized knowledge without navigating complex search processes [<xref ref-type="bibr" rid="ref8">8</xref>]. These features suggest potential utility in supporting clinical tasks that involve high information density and time constraints.</p><p>A growing body of research has explored the application of LLMs in medical contexts. Studies have demonstrated that models such as ChatGPT can achieve competitive performance in standardized medical licensing examinations, indicating a solid foundation in medical knowledge representation [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Other work has examined LLMs&#x2019; roles in patient education, clinical communication, and preliminary decision support, further supporting their potential utility in health care settings [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. However, existing evidence has largely focused on general medical scenarios, with relatively limited attention to complex nursing contexts, particularly oncology nursing. Compared with general medical tasks, oncology nursing involves more dynamic symptom trajectories, stricter safety thresholds, and a greater reliance on continuous assessment and timely intervention [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. These features place higher demands on both the accuracy and contextual adaptability of decision support tools.</p><p>In addition, most existing studies have focused on single-model evaluations, typically centered on ChatGPT, whereas the broader landscape of LLMs remains underexplored. In recent years, the development of LLMs has accelerated worldwide, accompanied by the rapid emergence of models developed in different linguistic and technological environments. These models differ not only in training data composition but also in architectural design and optimization strategies, which may influence their performance in clinically relevant tasks. This is particularly important because the performance of LLMs is closely associated with the linguistic distribution and domain characteristics of their training corpora. Previous studies have shown that models trained predominantly on English-language data may exhibit performance variability when applied to non&#x2013;English-language clinical contexts, where differences in language use, medical knowledge representation, and clinical practice patterns exist [<xref ref-type="bibr" rid="ref13">13</xref>]. Moreover, comparative research has demonstrated measurable differences between internationally developed models and those trained on Chinese-language corpora in medical tasks, suggesting that these variations are not solely due to language barriers but are also influenced by disparities in training data and model design [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Therefore, restricting evaluation to a single model provides only a limited understanding of current model capabilities. A comparative assessment across multiple models is necessary to capture the heterogeneity of performance and better reflect the range of technologies available in real-world clinical settings.</p><p>Beyond accuracy, other performance characteristics are also critical for clinical use. In nursing practice, decision support outputs must be readily interpretable, with sufficient clarity and conciseness to support timely decision-making while minimizing additional cognitive burden [<xref ref-type="bibr" rid="ref16">16</xref>]. In addition, the completeness of model-generated responses may influence usability in complex or information-dense clinical scenarios, particularly when tasks require extensive information processing or lengthy outputs. These factors may affect the practicality of integrating LLM outputs into routine clinical workflows [<xref ref-type="bibr" rid="ref17">17</xref>]. Standardized nursing examination questions, particularly case-based items, encapsulate key elements of clinical decision-making, including symptom progression, risk prioritization, and the timing of interventions [<xref ref-type="bibr" rid="ref14">14</xref>]. As such, they provide a structured and clinically relevant reference for evaluating the performance of LLMs in oncology nursing contexts. Therefore, this study aimed to compare the performance of multiple LLMs in oncology nursing decision support tasks using standardized examination questions and case-based clinical scenarios and explore LLMs&#x2019; potential applicability and current limitations in oncology nursing practice.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This study used a comparative cross-sectional design to evaluate the performance of multiple LLMs in oncology nursing decision support tasks. The evaluation was conducted as an observational in silico assessment focusing on the models&#x2019; ability to generate nursing-relevant responses to standardized clinical scenarios. No model retraining, fine-tuning, or parameter modification was performed. All models were assessed in their publicly accessible form, reflecting their real-world use by frontline nursing professionals. This study was reported with reference to relevant items from the STROBE (Strengthening the Reporting of Observational Studies in Epidemiology) statement to improve transparency in reporting study design, data sources, data collection procedures, outcome measures, and statistical analyses.</p></sec><sec id="s2-2"><title>Study Units and Setting</title><p>The primary study units were the responses generated by the evaluated LLMs. These responses were elicited using 2 types of question sets: case-based questions and standardized examination questions. The case-based questions were derived from a nationally used oncology nursing training manual developed by the oncology training center of our institution and published in 2021. This manual is routinely used in the training and assessment of oncology nurses and contains structured clinical scenarios representing common and complex situations in practice. All case-based questions contained in the manual were extracted in full and used in this study to ensure comprehensive coverage of the source material. Therefore, no a priori sample size calculation was performed as this study was based on the complete set of available standardized questions from the selected sources rather than a sampled subset. These scenarios require integration of symptom assessment, complication recognition, and clinical decision-making, thereby closely reflecting real-world oncology nursing practice. The examination questions were selected from a commercially published examination preparation book for the Chinese Nursing (Intermediate) Qualification Examination published by Liaoning Science and Technology Press in 2025. The resource contains standardized examination-oriented questions covering core nursing knowledge and competencies. In the present study, these questions were used as a standardized examination-based benchmark to compare model performance across commonly tested nursing knowledge domains.</p></sec><sec id="s2-3"><title>Data Collection</title><p>Data were collected by evaluating the performance of multiple LLMs on case-based questions and standardized examination questions. Five widely used generative AI chatbots were selected for comparison: DeepSeek, Qwen (Alibaba Cloud), Spark-Desk (iFlytek), WiseDiag, and ChatGPT (OpenAI). The selection aimed to include both internationally established and rapidly evolving Chinese-language LLMs as well as systems with different training characteristics and optimization strategies to capture variability in model performance across diverse technical and linguistic backgrounds. Case-based evaluations were conducted between October 2025 and November 2025 through the official public web interfaces of the evaluated models. The standardized examination assessment was completed in December 2025. All models were evaluated using the publicly available versions accessible to users at the time of data collection. Exact model version identifiers were not consistently displayed or available through the public web interfaces of all evaluated platforms. No attempts were made to access archived model versions or developer-specific configurations. ChatGPT was evaluated using the free web-based version available to the general public. DeepSeek was evaluated through its official web interface using the Fast Mode setting. Qwen, Spark-Desk, and WiseDiag were accessed through their respective official web-based platforms using the default configurations available to public users during the study period. No API access, model switching, parameter adjustment, or prompt optimization was performed. Each model was presented with identical questions using a standardized prompt: &#x201C;Please answer the following question. Provide only the final answer without explanation.&#x201D; The same prompt was applied to both case-based questions and standardized examination questions across all evaluated models. All questions were administered in a single-turn format. Each case set was initiated in a new session to minimize potential contextual influence from previous interactions. No additional clarification or follow-up prompts were provided. Following the approach described by Gilson et al [<xref ref-type="bibr" rid="ref18">18</xref>], each model was prompted once per question without repeated sampling. All responses were recorded verbatim for subsequent analysis.</p></sec><sec id="s2-4"><title>Measures</title><p>The performance of the LLMs was evaluated using both qualitative rating measures and objective examination-based indicators. For case-based questions, responses generated by each model were independently evaluated by 2 oncology nurses with more than 5 years of clinical experience. The raters were blinded to the identity of the models to minimize potential bias. Interrater reliability analyses were conducted using the initial independent ratings. The reported model performance scores were based on the mean ratings of the 2 raters. Response quality was assessed using 3 evaluation dimensions, namely, correctness, clarity, and conciseness. These dimensions have been frequently used in previous evaluations of LLM-generated responses in health care and medical education settings [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. In the present study, they were used to assess the accuracy, comprehensibility, and efficiency of information delivery in oncology nursing scenarios. Correctness referred to the accuracy of the content, clarity reflected the comprehensibility and coherence of the response, and conciseness indicated the efficiency of information delivery. Each dimension was rated on a 4-point Likert scale. A score of 4 indicated a completely correct, clear, or concise response; a score of 3 indicated a mostly correct, clear, or concise response; a score of 2 indicated a partially correct, clear, or concise response; and a score of 1 indicated a completely incorrect, unclear, or nonconcise response. For examination-based tasks, objective indicators included total score and accuracy rate. Consistent with the passing standard of the Nursing (Intermediate) Qualification Examination, a score of 60% was considered the minimum competency threshold.</p></sec><sec id="s2-5"><title>Statistical Analysis</title><p>Interrater reliability was assessed using quadratic weighted Cohen &#x03BA; for each of the 3 evaluation dimensions and a 2-way random-effects intraclass correlation coefficient (ICC) for the total correctness, clarity, and conciseness score. CIs for the weighted &#x03BA; values and ICCs were estimated using bootstrap resampling of paired ratings. To facilitate comparison with the original analysis, the Spearman rank correlation coefficient was also calculated between the 2 raters&#x2019; total scores. Agreement and disagreement patterns were further summarized using the proportions of exact agreement, 1-point differences, and differences of 2 points or more. All tests were 2-tailed, and a <italic>P</italic> value below .05 was considered statistically significant. Prism (version 10.1.2; GraphPad Software) was used for additional analyses. The Shapiro-Wilk test indicated that the data were not normally distributed; therefore, results were summarized as medians with IQRs. Differences among the 5 LLMs were evaluated using the Kruskal-Wallis test (2-sided; &#x03B1;=.05). When a significant overall difference was observed, pairwise comparisons were conducted using the Dunn post hoc test, with <italic>P</italic> values adjusted using the false discovery rate method to control for multiple testing. Adjusted <italic>P</italic> values below .05 were considered statistically significant.</p></sec><sec id="s2-6"><title>Ethical Considerations</title><p>This study primarily involved the evaluation of outputs generated by LLMs using standardized oncology nursing questions and case-based clinical scenarios. The study protocol was submitted to the Ethics Committee of Zhejiang Cancer Hospital and was determined to be exempt from full ethics review (exemption record number 202606291151000205900). The exemption was granted because this study did not involve patients, clinical interventions, identifiable personal data, sensitive personal information, or clinical decision-making activities. Institutional permission to use the 2021 oncology nursing training manual for research and publication purposes was obtained from the oncology training center of our institution. The objective assessment component was based on standardized nursing examination materials that did not contain patient-related, identifiable, or sensitive information. The subjective evaluation involved professional assessment of model-generated outputs by experienced oncology nurses using a predefined evaluation framework. All participating nurses were informed of the study purpose and evaluation procedures, and informed consent was obtained prior to participation.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Case-Based Evaluation</title><p>The weighted &#x03BA; values were 0.466 for correctness, 0.521 for clarity, and 0.453 for conciseness. The ICC for the total correctness, clarity, and conciseness score was 0.554. Across 165 model-generated responses, identical total scores were assigned in 72 (43.6%) cases. Differences of 1 point and 2 points or more were observed in 28.5% (n=47) and 27.9% (n=46) of cases, respectively. Detailed interrater agreement and disagreement statistics are shown in <xref ref-type="table" rid="table1">Table 1</xref>. For comparison with the original analysis, the Spearman rank correlation coefficient between raters was 0.566 (<italic>P</italic>&#x003C;.01). On the basis of 33 case-based questions derived from 10 clinical scenarios, the performance of the 5 LLMs is shown in <xref ref-type="table" rid="table2">Table 2</xref>. DeepSeek achieved the highest median score of 11.50 (IQR 10.50-12.00). Qwen and Spark-Desk obtained median scores of 11.00 (IQR 10.50-12.00) and 10.50 (IQR 9.50-11.50), respectively. WiseDiag and ChatGPT scored a median of 10.00 (IQR 9.50-11.50) and 10.00 (IQR 9.00-11.50), respectively. The Kruskal-Wallis test revealed a statistically significant difference in overall correctness, clarity, and conciseness scores among the 5 LLMs (<italic>H</italic>=11.416; <italic>P=</italic>.02). Post hoc analysis showed a significant difference only between DeepSeek and ChatGPT (raw <italic>P</italic>=.008; false discovery rate&#x2013;adjusted <italic>P</italic>=.04). No significant differences were found among the remaining comparisons.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Interrater reliability, agreement, and disagreement across the 3 evaluation dimensions (n=165 model-generated responses)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimension</td><td align="left" valign="bottom">Weighted &#x03BA; (95% CI)</td><td align="left" valign="bottom">Exact agreement, n (%)</td><td align="left" valign="bottom">1-point difference, n (%)</td><td align="left" valign="bottom">&#x2265;2-point difference, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Correctness</td><td align="left" valign="top">0.466 (0.305&#x2010;0.607)</td><td align="left" valign="top">103 (62.4)</td><td align="left" valign="top">50 (30.3)</td><td align="left" valign="top">12 (7.3)</td></tr><tr><td align="left" valign="top">Clarity</td><td align="left" valign="top">0.521 (0.341&#x2010;0.676)</td><td align="left" valign="top">122 (73.9)</td><td align="left" valign="top">29 (17.6)</td><td align="left" valign="top">14 (8.5)</td></tr><tr><td align="left" valign="top">Conciseness</td><td align="left" valign="top">0.453 (0.270&#x2010;0.614)</td><td align="left" valign="top">123 (74.5)</td><td align="left" valign="top">25 (15.2)</td><td align="left" valign="top">17 (10.3)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>The 2-way random-effects intraclass correlation coefficient for absolute agreement on the total correctness, clarity, and conciseness score was 0.554. </p></fn></table-wrap-foot></table-wrap><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Scores of the 5 large language models<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Questions, n</td><td align="left" valign="bottom">Score (range 3-12), median (IQR)</td></tr></thead><tbody><tr><td align="left" valign="top">DeepSeek</td><td align="left" valign="top">33</td><td align="left" valign="top">11.50 (10.50-12.00)<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">Qwen</td><td align="left" valign="top">33</td><td align="left" valign="top">11.00 (10.50-12.00)</td></tr><tr><td align="left" valign="top">Spark-Desk</td><td align="left" valign="top">33</td><td align="left" valign="top">10.50 (9.50-11.50)</td></tr><tr><td align="left" valign="top">WiseDiag</td><td align="left" valign="top">33</td><td align="left" valign="top">10.00 (9.50-11.50)</td></tr><tr><td align="left" valign="top">ChatGPT</td><td align="left" valign="top">33</td><td align="left" valign="top">10.00 (9.00-11.50)<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup><italic>H</italic>=11.416; <italic>P</italic>=.02.</p></fn><fn id="table2fn2"><p><sup>b</sup>The pairwise comparison between DeepSeek and ChatGPT was statistically significant (false discovery rate&#x2013;adjusted P=.04).</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Standardized Examination Assessment</title><p>Performance on the standardized examination questions is shown in <xref ref-type="table" rid="table3">Table 3</xref>. DeepSeek achieved the highest examination score at 93 and an accuracy of 93% (93/100). ChatGPT ranked second, scoring 88 with an accuracy of 88% (88/100). WiseDiag and Spark-Desk followed, with scores of 85 and 84, corresponding to accuracies of 85% (85/100) and 84% (84/100), respectively. Qwen showed the lowest performance, scoring 77 with an accuracy of 77% (77/100). According to the Nursing (Intermediate) Qualification Examination, all 5 evaluated LLMs achieved a passing level. Regarding completion of the full question set, only DeepSeek and ChatGPT completed all 100 questions within a single session. Qwen and WiseDiag required 2 sessions to provide responses for all questions, whereas Spark-Desk required 5 sessions during data collection. These observations describe the number of sessions required to obtain complete responses under the study conditions.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Performance of the 5 large language models on qualification examination questions.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Correct answers (score) in internal medicine (n=34), n (%)<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom">Correct answers (score) in surgery (n=22), n (%)<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="bottom">Correct answers (score) in gynecology (n=9), n (%)<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="bottom">Correct answers (score) in pediatrics (n=16), n (%)<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="bottom">Correct answers (score) in emergency and critical care (n=19), n (%)<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup></td><td align="left" valign="bottom" colspan="2">Overall (n=100)<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup></td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom">Sessions, n</td><td align="left" valign="bottom">Correct answers (score), n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">DeepSeek</td><td align="left" valign="top">30 (88.2)</td><td align="left" valign="top">21 (95.5)</td><td align="left" valign="top">9 (100)</td><td align="left" valign="top">14 (87.5)</td><td align="left" valign="top">19 (100)</td><td align="left" valign="top">1</td><td align="left" valign="top">93 (93)</td></tr><tr><td align="left" valign="top">Qwen</td><td align="left" valign="top">23 (67.6)</td><td align="left" valign="top">19 (86.4)</td><td align="left" valign="top">9 (100)</td><td align="left" valign="top">14 (87.5)</td><td align="left" valign="top">12 (63.2)</td><td align="left" valign="top">2</td><td align="left" valign="top">77 (77)</td></tr><tr><td align="left" valign="top">Spark-Desk</td><td align="left" valign="top">27 (79.4)</td><td align="left" valign="top">17 (77.3)</td><td align="left" valign="top">9 (100)</td><td align="left" valign="top">14 (87.5)</td><td align="left" valign="top">17 (89.5)</td><td align="left" valign="top">5</td><td align="left" valign="top">84 (84)</td></tr><tr><td align="left" valign="top">WiseDiag</td><td align="left" valign="top">28 (82.4)</td><td align="left" valign="top">19 (86.4)</td><td align="left" valign="top">8 (88.9)</td><td align="left" valign="top">13 (81.3)</td><td align="left" valign="top">17 (89.5)</td><td align="left" valign="top">2</td><td align="left" valign="top">85 (85)</td></tr><tr><td align="left" valign="top">ChatGPT</td><td align="left" valign="top">29 (85.3)</td><td align="left" valign="top">19 (86.4)</td><td align="left" valign="top">8 (88.9)</td><td align="left" valign="top">14 (87.5)</td><td align="left" valign="top">18 (94.7)</td><td align="left" valign="top">1</td><td align="left" valign="top">88 (88)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Mean correct answers 27.4 (SD 2.7); mean accuracy 80.6% (SD 7.9%).</p></fn><fn id="table3fn2"><p><sup>b</sup>Mean correct answers 19.0 (SD 1.4); mean accuracy 86.4% (SD 6.4%).</p></fn><fn id="table3fn3"><p><sup>c</sup>Mean correct answers 8.6 (SD 0.5); mean accuracy 95.6% (SD 6.1%).</p></fn><fn id="table3fn4"><p><sup>d</sup>Mean correct answers 13.8 (SD 0.4); mean accuracy 86.3% (SD 2.8%).</p></fn><fn id="table3fn5"><p><sup>e</sup>Mean correct answers 16.6 (SD 2.7); mean accuracy 87.4% (SD 14.2%).</p></fn><fn id="table3fn6"><p><sup>f</sup>Mean sessions 2.2 (SD 1.6); mean correct answers 85.4 (SD 5.9); mean accuracy 85.4% (SD 5.9%).</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Main Findings</title><p>The present study compared the performance of 5 widely used LLMs in oncology nursing decision support tasks using both case-based clinical scenarios and standardized examination questions. Overall, all evaluated models achieved passing scores on the examination assessment and obtained median correctness, clarity, and conciseness scores ranging from 10.00 to 11.50 in the case-based evaluation. DeepSeek achieved the highest score in the case-based assessment, whereas ChatGPT ranked second in the examination-based evaluation. These findings suggest that model performance may vary according to the type of task being evaluated. Examination performance also varied across knowledge domains, with relatively lower accuracy observed in internal medicine and emergency and critical care questions compared with other examination categories. Taken together, the findings suggest that current LLMs perform relatively well on structured nursing knowledge assessments, although performance remains variable across different task types and clinical domains.</p></sec><sec id="s4-2"><title>Error Patterns and Clinical Reasoning Challenges</title><p>Review of incorrect responses suggested that some challenges may have been associated with scenarios involving oncology-specific nursing knowledge, complex clinical reasoning, emergency management, and threshold-based clinical decisions. Because these observations were derived from qualitative review rather than a predefined error classification framework, they should be interpreted cautiously. Nevertheless, they may provide useful insights into clinical situations that remain challenging for current LLMs [<xref ref-type="bibr" rid="ref21">21</xref>]. Tofeeq et al [<xref ref-type="bibr" rid="ref9">9</xref>] reported that LLMs generally achieve high accuracy in basic medical question&#x2013;answering tasks but their performance is uneven across different task dimensions, particularly in complex reasoning and highly specialized domains. These findings suggest that the limitations of LLMs in oncology nursing may not be related only to knowledge coverage. Rather, they may also reflect the difficulty of applying general knowledge to context-sensitive clinical situations. Oncology nursing requires not only factual knowledge but also the integration of dynamic clinical information, including symptom progression, treatment stage, and patient-specific conditions. Similar limitations have been reported in other oncology-related applications of LLMs, where performance may decline when tasks require interpretation and management of complex treatment-related information [<xref ref-type="bibr" rid="ref22">22</xref>]. Clinical scenarios involving urgent decision-making or interpretation of clinical thresholds may be particularly difficult for current LLMs. Such situations require accurate interpretation of patient information and timely prioritization of actions. Previous systematic reviews have highlighted limitations related to knowledge coverage, contextual adaptability, and safety [<xref ref-type="bibr" rid="ref23">23</xref>]. In the present study, difficulties appeared more evident in scenarios involving case analysis, integrative reasoning, numerical thresholds, and contraindication judgments. Some model-generated responses included clinically inappropriate recommendations in these scenarios even when they appeared logically coherent. This limitation has also been reported in previous evaluations of LLMs in health care settings [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. One possible explanation is that LLMs generate responses based on probabilistic language patterns and may therefore produce recommendations that appear plausible but lack the precision required for clinical decision-making. This may partly explain the observed difficulties in clinically complex scenarios despite generally acceptable performance on structured knowledge-based tasks [<xref ref-type="bibr" rid="ref1">1</xref>].</p><p>These observations also point to the distinctive demands of nursing decision-making. Unlike diagnosis-oriented tasks, nursing practice requires continuous patient monitoring, risk prioritization, and individualized interventions. Therefore, evaluation of LLMs in nursing should prioritize context sensitivity, temporal dynamics, and risk stratification. These processes are highly context dependent and require integration of multiple sources of clinical information. In oncology nursing, this complexity is further amplified by rapidly evolving oncologic situations and treatment-related toxicity profiles [<xref ref-type="bibr" rid="ref2">2</xref>]. As a result, the limitations of LLMs extend beyond incomplete knowledge coverage to include difficulties in translating knowledge into actionable nursing decisions. This aligns with previous research indicating that LLMs may produce outputs that are linguistically coherent but not always clinically appropriate [<xref ref-type="bibr" rid="ref24">24</xref>]. From a practical perspective, current LLMs may be better suited to supportive functions such as information retrieval and patient education than to independent clinical decision-making in oncology nursing settings.</p></sec><sec id="s4-3"><title>Differences Across LLMs</title><p>In terms of model comparison, DeepSeek achieved the highest score in the case-based evaluation, whereas the remaining models showed comparatively lower performance. This pattern differs from some findings reported in English-language settings. Differences in language environment, training data distribution, and domain-specific corpus coverage may partly explain the variation in model performance [<xref ref-type="bibr" rid="ref6">6</xref>]. Consistent with this interpretation, Lai et al [<xref ref-type="bibr" rid="ref26">26</xref>] reported that LLM performance is highly dependent on the language structure and domain coverage of the training corpus.</p><p>Models with stronger Chinese-language and domain-specific training may therefore have been better able to capture local clinical terminology and practice patterns in Chinese nursing scenarios, which is broadly consistent with findings reported in studies of Chinese medical LLMs [<xref ref-type="bibr" rid="ref27">27</xref>]. An additional observation was the discrepancy in ChatGPT&#x2019;s performance across the 2 evaluation tasks. Although ChatGPT achieved the second-highest accuracy in the standardized examination assessment, its performance in the case-based correctness, clarity, and conciseness evaluation was comparatively lower. This observation suggests that performance on structured examination questions may not necessarily translate into stronger performance in case-based nursing scenarios. Standardized examination questions tend to emphasize recognition of predefined answers and structured knowledge application. In contrast, case-based questions require integration of contextual information, prioritization of nursing problems, and application of knowledge within dynamic clinical situations. These tasks place different cognitive demands on LLMs. Therefore, the observed discrepancy may reflect differences in task characteristics rather than differences in knowledge representation alone. These results indicate that examination performance alone may not adequately reflect model performance in clinically contextualized nursing scenarios [<xref ref-type="bibr" rid="ref9">9</xref>]. This discrepancy highlights the value of including clinically contextualized case scenarios when evaluating LLM performance in nursing. Although DeepSeek scored significantly higher than ChatGPT, no significant differences were observed among the remaining pairwise comparisons after adjustment for multiple testing. The relatively small number of case-based questions, together with correction for multiple comparisons, may have limited the ability to detect smaller performance differences between models. Therefore, the absence of statistical significance should not be interpreted as evidence of equivalent performance. Larger and more diverse question sets may help determine whether additional performance differences exist among models. Consistent with previous studies, the present findings suggest that current LLMs retain important limitations when applied to complex clinical reasoning tasks [<xref ref-type="bibr" rid="ref9">9</xref>]. Moreover, none of the evaluated models achieved a consistently error-free level of performance in the present study. Differences in the number of sessions required to obtain complete responses were also observed across the evaluated platforms. Because all models were accessed through publicly available web-based interfaces, these observations may reflect characteristics of the web-based platforms in addition to the models themselves. However, the present study was not designed to evaluate platform-level factors, and the reasons for these differences remain unclear. Accordingly, the number of sessions required to obtain complete responses should not be interpreted as a direct measure of intrinsic model capability.</p></sec><sec id="s4-4"><title>Interpretation of Correctness, Clarity, and Conciseness Scores and Rating Reliability</title><p>Interpretation of the correctness, clarity, and conciseness evaluation results should take the prompting strategy into account. All models were instructed to provide concise final answers without explanation, and the same prompt was applied across both case-based and examination questions. Gilson et al [<xref ref-type="bibr" rid="ref18">18</xref>] adopted a standardized prompting strategy when evaluating LLM performance. Similar evaluation approaches have been adopted in studies assessing LLM performance in health care settings, where evaluation has primarily focused on generated outputs [<xref ref-type="bibr" rid="ref9">9</xref>]. Therefore, the clarity and conciseness dimensions should be interpreted as indicators of final response quality under a standardized prompting condition rather than as direct assessments of detailed clinical reasoning, reasoning transparency, or broader nursing communication ability [<xref ref-type="bibr" rid="ref28">28</xref>]. Future evaluations using case-based nursing scenarios may benefit from prompts designed to elicit step-by-step reasoning when reasoning transparency is the outcome of interest. A separate issue concerns the reliability of the subjective evaluation process. Using the Landis and Koch interpretation scale for agreement coefficients, the weighted &#x03BA; values for correctness, clarity, and conciseness fell within the moderate range. The ICC for the total correctness, clarity, and conciseness score also indicated moderate reliability. Although these results suggest an acceptable level of agreement for an exploratory evaluation, some variability remained between evaluators. This may partly reflect the subjective nature of assessing response quality, particularly for dimensions such as clarity and conciseness, which are inherently more open to interpretation than factual correctness. Hallgren [<xref ref-type="bibr" rid="ref29">29</xref>] noted that variability between raters is common when subjective judgment is involved even when standardized evaluation criteria are used. Therefore, the subjective evaluation results should be interpreted with appropriate caution. Given the subjective nature of some assessment dimensions, the use of more detailed scoring rubrics and structured rater calibration procedures may help improve consistency in future evaluations [<xref ref-type="bibr" rid="ref29">29</xref>].</p></sec><sec id="s4-5"><title>Implications for Oncology Nursing Practice</title><p>The findings suggest that LLMs may have practical value in oncology nursing when used for structured information processing, educational support, and routine communication tasks. Song et al [<xref ref-type="bibr" rid="ref23">23</xref>] reported that LLMs can achieve relatively high accuracy in standardized medical examinations, particularly in structured question-answering tasks. Will et al [<xref ref-type="bibr" rid="ref30">30</xref>] further showed that LLMs may improve the readability and comprehension of complex medical information. These findings support the potential use of LLMs as supplementary tools in nursing education and patient education. In oncology nursing, such tools may help organize complex information, summarize standardized care protocols, and prepare patient-facing educational materials. The integration of LLMs into oncology nursing practice raises important considerations regarding patient safety, ethical responsibility, and accountability. Because nursing decisions may directly affect patient outcomes, clear frameworks for risk management and responsibility attribution will be necessary before broader clinical implementation can be considered [<xref ref-type="bibr" rid="ref31">31</xref>]. In clinical practice, LLMs may be most appropriately integrated into specific components of the nursing workflow rather than the entire clinical decision-making process. They may assist with organizing symptom-related information and summarizing risk factors during patient assessment, supporting structured documentation of treatment-related adverse events during treatment, and generating discharge instructions or self-care guidance for patient education. These functions primarily support information organization and communication rather than direct patient care decisions. With appropriate governance frameworks and professional oversight, LLMs may contribute to more efficient clinical information management and documentation processes in oncology nursing settings. However, their application should remain clearly bounded within a supportive rather than autonomous role. Particular caution may be required in complex oncology nursing situations, such as toxicity monitoring, infection risk assessment, and dynamic symptom management during cancer treatment. These scenarios often require integration of patient-specific information and clinical context, which may present challenges for current LLMs. Future studies should focus on domain-specific fine-tuning using oncology nursing guidelines and clinical pathways together with validation in real-world clinical environments. Such work may help clarify the role of LLMs in oncology nursing practice and support their safe and effective implementation.</p></sec><sec id="s4-6"><title>Limitations</title><p>This study has several limitations. Although all eligible case-based questions from the selected training manual were included, the overall number of questions remained limited. This may have reduced statistical power for pairwise comparisons and made it harder to detect smaller performance differences among models, particularly after adjustment for multiple testing. The evaluation was based on standardized test items, which ensured comparability across models but may not fully reflect performance in real-world clinical settings characterized by multimorbidity, dynamic disease progression, and incomplete information. Although oncology-related content was included, highly complex scenarios specific to oncology nursing, such as cross-system coordination and long-term care decision-making, were underrepresented. The examination assessment relied on a commercially published preparation resource rather than official examinations. Although the resource was developed for preparation for the Chinese Nursing (Intermediate) Qualification Examination, its representativeness of current examination standards cannot be fully guaranteed. Potential overlap between some evaluation items and materials that may have been accessible during model training also cannot be excluded. Because the training data of the evaluated LLMs are not publicly available, the extent of possible memorization could not be assessed directly. Several methodological factors should also be considered. Differences in training data, reasoning strategies, and sensitivity to question phrasing may also have influenced the results. A uniform prompt was used to ensure comparability across models, but this approach may have underestimated performance under optimized prompting conditions. Each model was queried only once for each question, and repeated sampling was not performed. Response variability across repeated runs was therefore not assessed. In addition, specific model version identifiers could not be retrospectively verified for all platforms. Updates to publicly available models during the data collection period cannot be completely ruled out, and their potential influence on model performance could not be assessed directly. This study did not include a human comparator group, which limits direct comparison between model performance and clinical practice. Regional differences in health care systems, clinical guidelines, and cultural contexts were also not considered and may affect the generalizability of the findings.</p></sec><sec id="s4-7"><title>Conclusions</title><p>All evaluated LLMs achieved passing scores on the standardized examination assessment, but performance varied across case-based oncology nursing scenarios. DeepSeek achieved the highest performance in the case-based evaluation, whereas performance on examination questions did not consistently correspond to performance in clinically contextualized assessments. Limitations were most evident in scenarios requiring clinical reasoning, contextual interpretation, and risk-sensitive decision-making. These findings suggest that performance on standardized examination questions alone may not adequately reflect the ability of LLMs to support nursing practice in real-world clinical settings. These findings indicate that examination-style testing may be complemented by clinically contextualized case scenarios when evaluating LLMs in nursing. From a practical perspective, current LLMs may be better suited to tasks such as information retrieval, knowledge organization, and patient education than to independent clinical decision-making. Further progress in this area will require domain-specific development and rigorous clinical validation. Appropriate governance frameworks will also be essential for the safe integration of LLMs into oncology nursing practice.</p></sec></sec></body><back><ack><p>The authors thank the oncology nurses who participated in the evaluation for their time and professional expertise in assessing the model-generated responses. The authors declare the use of generative AI (GenAI) in the research and writing process. According to the 2025 Generative AI Delegation Taxonomy, the following task was delegated to GenAI tools under full human supervision: translation. The GenAI tool used was ChatGPT (OpenAI). Responsibility for the final manuscript lies entirely with the authors. GenAI tools are not listed as authors and do not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>This study was supported by the 2024 Zhejiang Provincial Basic Public Welfare Research Project (LTGY24H160028). The funder had no involvement in the study design, data collection and analysis, interpretation of the results, or writing of the manuscript.</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed during this study, including model-generated responses and rating data, are available from the corresponding author on reasonable request. The source materials used to construct the evaluation questions, including standardized examination questions and case-based materials derived from the 2021 oncology nursing training manual, are not publicly available due to copyright restrictions.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: QZ, WW</p><p>Data curation: QZ, DH, XC</p><p>Formal analysis: QZ, YX</p><p>Investigation: QZ</p><p>Methodology: QZ, WW</p><p>Supervision: WW</p><p>Validation: YJ, HH</p><p>Writing&#x2014;original draft: QZ</p><p>Writing&#x2014;review and editing: QZ, YX, WW</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb2">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb3">STROBE</term><def><p>Strengthening the Reporting of Observational Studies in Epidemiology</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scott&#x00E9;</surname><given-names>F</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>A</given-names> </name><name name-style="western"><surname>Davies</surname><given-names>A</given-names> </name></person-group><article-title>Supportive care: the &#x201C;keystone&#x201D; of modern oncology practice</article-title><source>Cancers (Basel)</source><year>2023</year><month>07</month><day>29</day><volume>15</volume><issue>15</issue><fpage>3860</fpage><pub-id pub-id-type="doi">10.3390/cancers15153860</pub-id><pub-id pub-id-type="medline">37568675</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maguire</surname><given-names>R</given-names> </name><name name-style="western"><surname>McCann</surname><given-names>L</given-names> </name><name name-style="western"><surname>Kotronoulas</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Real time remote symptom monitoring during chemotherapy for cancer: European multicentre randomised controlled trial (eSMART)</article-title><source>BMJ</source><year>2021</year><month>07</month><day>21</day><volume>374</volume><fpage>n1647</fpage><pub-id pub-id-type="doi">10.1136/bmj.n1647</pub-id><pub-id pub-id-type="medline">34289996</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gardner</surname><given-names>C</given-names> </name><name name-style="western"><surname>Halligan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Fontana</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Evaluation of a clinical decision support tool for matching cancer patients to clinical trials using simulation-based research</article-title><source>Health Informatics J</source><year>2022</year><volume>28</volume><issue>2</issue><fpage>14604582221087890</fpage><pub-id pub-id-type="doi">10.1177/14604582221087890</pub-id><pub-id pub-id-type="medline">35450483</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chow</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name></person-group><article-title>Large language models in medical chatbots: opportunities, challenges, and the need to address AI risks</article-title><source>Information</source><year>2025</year><month>06</month><day>27</day><volume>16</volume><issue>7</issue><fpage>549</fpage><pub-id pub-id-type="doi">10.3390/info16070549</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tian</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Yeganova</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Opportunities and challenges for ChatGPT and large language models in biomedicine and health</article-title><source>Brief Bioinform</source><year>2023</year><month>11</month><day>22</day><volume>25</volume><issue>1</issue><fpage>bbad493</fpage><pub-id pub-id-type="doi">10.1093/bib/bbad493</pub-id><pub-id pub-id-type="medline">38168838</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Siam</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Varela</surname><given-names>A</given-names> </name><name name-style="western"><surname>Faruk</surname><given-names>MJ</given-names> </name><etal/></person-group><article-title>Benchmarking large language models on the United States Medical Licensing Examination for clinical reasoning and medical licensing scenarios</article-title><source>Sci Rep</source><year>2025</year><month>12</month><day>3</day><volume>16</volume><issue>1</issue><fpage>1387</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-31010-4</pub-id><pub-id pub-id-type="medline">41339739</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ganjavi</surname><given-names>C</given-names> </name><name name-style="western"><surname>Eppler</surname><given-names>M</given-names> </name><name name-style="western"><surname>O&#x2019;Brien</surname><given-names>D</given-names> </name><etal/></person-group><article-title>ChatGPT and large language models (LLMs) awareness and use. A prospective cross-sectional survey of U.S. medical students</article-title><source>PLOS Digit Health</source><year>2024</year><month>09</month><volume>3</volume><issue>9</issue><fpage>e0000596</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000596</pub-id><pub-id pub-id-type="medline">39236008</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>YG</given-names> </name></person-group><article-title>Exploratory search with generative AI: an empirical study on the impact of interaction design strategies on information exploration and cognitive load</article-title><source>Int J Hum Comput Stud</source><year>2026</year><month>03</month><volume>210</volume><fpage>103771</fpage><pub-id pub-id-type="doi">10.1016/j.ijhcs.2026.103771</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tofeeq</surname><given-names>K</given-names> </name><name name-style="western"><surname>Naseer</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wali</surname><given-names>A</given-names> </name></person-group><article-title>Large language models in healthcare: a systematic evaluation on medical Q/A datasets</article-title><source>Health Inf Sci Syst</source><year>2025</year><volume>14</volume><issue>1</issue><fpage>2</fpage><pub-id pub-id-type="doi">10.1007/s13755-025-00397-9</pub-id><pub-id pub-id-type="medline">41281608</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Di Battista</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kernitsky</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dibart</surname><given-names>S</given-names> </name></person-group><article-title>Artificial intelligence chatbots in patient communication: current possibilities</article-title><source>Int J Periodontics Restorative Dent</source><year>2024</year><month>11</month><day>15</day><volume>44</volume><issue>6</issue><fpage>731</fpage><lpage>738</lpage><pub-id pub-id-type="doi">10.11607/prd.6925</pub-id><pub-id pub-id-type="medline">37819844</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Li</surname><given-names>G</given-names> </name></person-group><article-title>ChatGPT for mechanobiology and medicine: a perspective</article-title><source>Mechanobiol Med</source><year>2023</year><volume>1</volume><issue>1</issue><fpage>100005</fpage><pub-id pub-id-type="doi">10.1016/j.mbm.2023.100005</pub-id><pub-id pub-id-type="medline">40395874</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nicholson</surname><given-names>B</given-names> </name><name name-style="western"><surname>Sloss</surname><given-names>EA</given-names> </name><name name-style="western"><surname>Smiley</surname><given-names>A</given-names> </name><name name-style="western"><surname>Finkelstein</surname><given-names>J</given-names> </name><name name-style="western"><surname>Mooney</surname><given-names>K</given-names> </name></person-group><article-title>Perception of AI symptom models in oncology nursing: mixed methods evaluation study</article-title><source>JMIR Nurs</source><year>2026</year><month>02</month><day>4</day><volume>9</volume><fpage>e82283</fpage><pub-id pub-id-type="doi">10.2196/82283</pub-id><pub-id pub-id-type="medline">41637487</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Tsang</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name></person-group><article-title>Performance unfairness of large language models in cross-language fact-checking</article-title><source>Inf Process Manag</source><year>2026</year><volume>63</volume><issue>4</issue><fpage>104616</fpage><pub-id pub-id-type="doi">10.1016/j.ipm.2026.104616</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Strasser</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Anschuetz</surname><given-names>W</given-names> </name><name name-style="western"><surname>Dennst&#x00E4;dt</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hastings</surname><given-names>J</given-names> </name></person-group><article-title>Performance evaluation of large language models in multilingual medical multiple-choice questions: mixed methods study</article-title><source>JMIR Med Educ</source><year>2026</year><month>03</month><day>5</day><volume>12</volume><fpage>e81399</fpage><pub-id pub-id-type="doi">10.2196/81399</pub-id><pub-id pub-id-type="medline">41813244</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Duan</surname><given-names>L</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chi</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sheng</surname><given-names>D</given-names> </name></person-group><article-title>Performance of large language models in the non-English context: qualitative study of models trained on different languages in Chinese medical examinations</article-title><source>JMIR Med Inform</source><year>2025</year><month>06</month><day>27</day><volume>13</volume><fpage>e69485</fpage><pub-id pub-id-type="doi">10.2196/69485</pub-id><pub-id pub-id-type="medline">40577654</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Olawade</surname><given-names>DB</given-names> </name><name name-style="western"><surname>Clement David-Olawade</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rotifa</surname><given-names>OB</given-names> </name><name name-style="western"><surname>Wada</surname><given-names>OZ</given-names> </name></person-group><article-title>Artificial intelligence in Nigerian nursing education: are future nurses prepared for the digital revolution in healthcare?</article-title><source>Nurse Educ Pract</source><year>2025</year><month>08</month><volume>87</volume><fpage>104511</fpage><pub-id pub-id-type="doi">10.1016/j.nepr.2025.104511</pub-id><pub-id pub-id-type="medline">40819547</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nazi</surname><given-names>ZA</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>W</given-names> </name></person-group><article-title>Large language models in healthcare and medical domain: a review</article-title><source>Informatics</source><year>2024</year><volume>11</volume><issue>3</issue><fpage>57</fpage><pub-id pub-id-type="doi">10.3390/informatics11030057</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gilson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Safranek</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>How does ChatGPT perform on the United States Medical Licensing Examination (USMLE)? The implications of large language models for medical education and knowledge assessment</article-title><source>JMIR Med Educ</source><year>2023</year><month>02</month><day>8</day><volume>9</volume><fpage>e45312</fpage><pub-id pub-id-type="doi">10.2196/45312</pub-id><pub-id pub-id-type="medline">36753318</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aggarwal</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tam</surname><given-names>CC</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><name name-style="western"><surname>Qiao</surname><given-names>S</given-names> </name></person-group><article-title>Artificial intelligence-based chatbots for promoting health behavioral changes: systematic review</article-title><source>J Med Internet Res</source><year>2023</year><month>02</month><day>24</day><volume>25</volume><fpage>e40789</fpage><pub-id pub-id-type="doi">10.2196/40789</pub-id><pub-id pub-id-type="medline">36826990</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yaneva</surname><given-names>V</given-names> </name><name name-style="western"><surname>Baldwin</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jurich</surname><given-names>DP</given-names> </name><name name-style="western"><surname>Swygert</surname><given-names>K</given-names> </name><name name-style="western"><surname>Clauser</surname><given-names>BE</given-names> </name></person-group><article-title>Examining ChatGPT performance on USMLE sample items and implications for assessment</article-title><source>Acad Med</source><year>2024</year><month>02</month><day>1</day><volume>99</volume><issue>2</issue><fpage>192</fpage><lpage>197</lpage><pub-id pub-id-type="doi">10.1097/ACM.0000000000005549</pub-id><pub-id pub-id-type="medline">37934828</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cahyawijaya</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>N</given-names> </name><etal/></person-group><article-title>A multitask, multilingual, multimodal evaluation of ChatGPT on reasoning, hallucination, and interactivity</article-title><source>Proceedings of the 13th International Joint Conference on Natural Language Processing and the 3rd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>675</fpage><lpage>718</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.ijcnlp-main.45</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zitu</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Manne</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Rahat</surname><given-names>WB</given-names> </name><name name-style="western"><surname>Binkheder</surname><given-names>S</given-names> </name></person-group><article-title>Large language models for drug-related adverse events in oncology pharmacy: detection, grading, and actioning</article-title><source>Pharmacy (Basel)</source><year>2025</year><month>12</month><day>3</day><volume>13</volume><issue>6</issue><fpage>176</fpage><pub-id pub-id-type="doi">10.3390/pharmacy13060176</pub-id><pub-id pub-id-type="medline">41441324</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Application and challenges of large language models in clinical nursing: a systematic review</article-title><source>Comput Inform Nurs</source><year>2025</year><month>09</month><day>1</day><volume>43</volume><issue>9</issue><fpage>e01328</fpage><pub-id pub-id-type="doi">10.1097/CIN.0000000000001328</pub-id><pub-id pub-id-type="medline">40526735</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reddy</surname><given-names>S</given-names> </name></person-group><article-title>Evaluating large language models for use in healthcare: a framework for translational value assessment</article-title><source>Inform Med Unlocked</source><year>2023</year><volume>41</volume><fpage>101304</fpage><pub-id pub-id-type="doi">10.1016/j.imu.2023.101304</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>J</given-names> </name></person-group><article-title>Can large language models facilitate the effective implementation of nursing processes in clinical settings?</article-title><source>BMC Nurs</source><year>2025</year><month>04</month><day>8</day><volume>24</volume><issue>1</issue><fpage>394</fpage><pub-id pub-id-type="doi">10.1186/s12912-025-03010-2</pub-id><pub-id pub-id-type="medline">40200247</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lai</surname><given-names>V</given-names> </name><name name-style="western"><surname>Ngo</surname><given-names>N</given-names> </name><name name-style="western"><surname>Pouran Ben Veyseh</surname><given-names>A</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Bouamor</surname><given-names>H</given-names> </name><name name-style="western"><surname>Pino</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bali</surname><given-names>K</given-names> </name></person-group><article-title>ChatGPT beyond English: towards a comprehensive evaluation of large language models in multilingual learning</article-title><source>Findings of the Association for Computational Linguistics: EMNLP 2023</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>13171</fpage><lpage>13189</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.findings-emnlp.878</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Tian</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gan</surname><given-names>R</given-names> </name><name name-style="western"><surname>Song</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Ku</surname><given-names>LW</given-names> </name><name name-style="western"><surname>Martins</surname><given-names>A</given-names> </name><name name-style="western"><surname>Srikumar</surname><given-names>V</given-names> </name></person-group><article-title>ChiMed-GPT: a Chinese medical large language model with full training regime and better alignment to human preferences</article-title><source>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics</source><year>2024</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>7156</fpage><lpage>7173</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.acl-long.386</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Bubeck</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chandrasekaran</surname><given-names>V</given-names> </name><name name-style="western"><surname>Eldan</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Sparks of artificial general intelligence: early experiments with GPT-4</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 22, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.12712</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hallgren</surname><given-names>KA</given-names> </name></person-group><article-title>Computing inter-rater reliability for observational data: an overview and tutorial</article-title><source>Tutor Quant Methods Psychol</source><year>2012</year><volume>8</volume><issue>1</issue><fpage>23</fpage><lpage>34</lpage><pub-id pub-id-type="doi">10.20982/tqmp.08.1.p023</pub-id><pub-id pub-id-type="medline">22833776</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Will</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zaretsky</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dowlath</surname><given-names>A</given-names> </name><name name-style="western"><surname>Testa</surname><given-names>P</given-names> </name><name name-style="western"><surname>Feldman</surname><given-names>J</given-names> </name></person-group><article-title>Enhancing the readability of online patient education materials using large language models: cross-sectional study</article-title><source>J Med Internet Res</source><year>2025</year><month>06</month><day>4</day><volume>27</volume><fpage>e69955</fpage><pub-id pub-id-type="doi">10.2196/69955</pub-id><pub-id pub-id-type="medline">40465378</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Topol</surname><given-names>EJ</given-names> </name></person-group><article-title>High-performance medicine: the convergence of human and artificial intelligence</article-title><source>Nat Med</source><year>2019</year><month>01</month><volume>25</volume><issue>1</issue><fpage>44</fpage><lpage>56</lpage><pub-id pub-id-type="doi">10.1038/s41591-018-0300-7</pub-id><pub-id pub-id-type="medline">30617339</pub-id></nlm-citation></ref></ref-list></back></article>