<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e89858</article-id><article-id pub-id-type="doi">10.2196/89858</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Comparative Performance of AI Models and Clinicians in Evidence-Based Cardiovascular Disease Management for People Living With HIV: Comparative Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Kong</surname><given-names>Tianqi</given-names></name><degrees>MPH</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sun</surname><given-names>Liqin</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Luo</surname><given-names>Yinsong</given-names></name><degrees>MPH</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Xiao</surname><given-names>Xi</given-names></name><degrees>MPH</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Jin</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Liu</surname><given-names>Jiaye</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>School of Public Health, Shenzhen University Medical School, Shenzhen University</institution><addr-line>No.1066 Xueyuan Avenue</addr-line><addr-line>Shenzhen</addr-line><addr-line>Guangdong</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Infectious Diseases, National Clinical Research Center for Infectious Diseases, Shenzhen Third People&#x2019;s Hospital</institution><addr-line>Shenzhen</addr-line><addr-line>Guangdong</addr-line><country>China</country></aff><aff id="aff3"><institution>Department of Infectious Diseases, The Ninth People's Hospital of Dongguan</institution><addr-line>Dongguan</addr-line><addr-line>Guangdong</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>AL-Asadi</surname><given-names>Ali</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Chang</surname><given-names>Larry</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Jiaye Liu, MD, PhD, School of Public Health, Shenzhen University Medical School, Shenzhen University, No.1066 Xueyuan Avenue, Shenzhen, Guangdong, 518060, China, 86 18819026906; <email>liujiaye1984@163.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>10</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e89858</elocation-id><history><date date-type="received"><day>18</day><month>12</month><year>2025</year></date><date date-type="rev-recd"><day>22</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>13</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Tianqi Kong, Liqin Sun, Yinsong Luo, Xi Xiao, Jin Li, Jiaye Liu. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 10.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e89858"/><abstract><sec><title>Background</title><p>Although widespread antiretroviral therapy has extended the life expectancy of people living with HIV, cardiovascular disease (CVD) has emerged as a primary comorbidity. Persistent cross-specialty knowledge gaps in routine clinical practice lead to suboptimal adherence to guidelines. Integrated, evidence-based tools are urgently needed to overcome these interdisciplinary barriers. While large language models (LLMs) have demonstrated significant capabilities in medicine, no systematic evaluation has assessed their ability to facilitate multidisciplinary CVD management for people living with HIV.</p></sec><sec><title>Objective</title><p>This study compared the performance of 4 mainstream AI models (DeepSeek-V3, DeepSeek-R1, ChatGPT-4o, and ChatGPT-o4-mini) against 12 human clinicians (8 infectious disease specialists and 4 cardiologists) in addressing guideline-based CVD management tasks for people living with HIV.</p></sec><sec sec-type="methods"><title>Methods</title><p>Based on 4 authoritative domestic and international HIV/CVD guidelines, a structured 25-question assessment was developed via 2 rounds of Delphi consultation. Standard reference answers and an evaluation framework were finalized through expert consensus. LLM responses were generated using standardized prompts. Clinicians answered identical questions via one-on-one structured interviews, transcribed verbatim. Six multidisciplinary experts independently rated all responses across 4 dimensions&#x2014;accuracy, completeness, readability, and reliability&#x2014;using a 4-point ordinal scale (1=poor to 4=excellent). Cumulative link mixed models analyzed intergroup differences.</p></sec><sec sec-type="results"><title>Results</title><p>All AI models achieved significantly higher scores than clinicians across all dimensions (<italic>P</italic>&#x003C;.001). The AI group&#x2019;s mean scores ranged from 3.44 to 3.68 (median 4, IQR 3.0-4.0; coefficient of variation=0.145-0.178). Conversely, clinicians&#x2019; scores were lower (mean 1.78-2.05; median 2, IQR 1.0-3.0; coefficient of variation=0.428-0.473) with marked dispersion. DeepSeek-R1 delivered the optimal performance, significantly outperforming the other 3 models (all <italic>P</italic>&#x003C;.001). Specialty-stratified analysis revealed no significant overall score difference between cardiologists and infectious disease specialists (odds ratio 0.92, 95% CI 0.84-1.01; <italic>P</italic>=.09). However, dimension-specific analysis indicated that cardiologists scored higher in accuracy (odds ratio 0.81, 95% CI 0.67-0.97; <italic>P</italic>=.03). Domain-specific divergence was evident: cardiologists outperformed infectious disease specialists in CVD risk assessment (2.26 vs 1.83), whereas infectious disease specialists led in drug adverse effect evaluation (2.23 vs 1.65).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In this structured question-and-answer study, LLMs outperformed human clinicians across all metrics for HIV-associated CVD management, with DeepSeek-R1 achieving superior composite scores. These findings validate DeepSeek-R1&#x2019;s potential as a cross-disciplinary decision-support tool capable of integrating complex clinical knowledge, mitigating specialty gaps, and enhancing information precision. Integrating AI systems into multidisciplinary workflows, complemented by targeted clinical training, may optimize the management of complex comorbidities in people living with HIV.</p></sec></abstract><kwd-group><kwd>HIV</kwd><kwd>cardiovascular disease</kwd><kwd>AI</kwd><kwd>comorbidity management</kwd><kwd>patient education</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>HIV infection remains a major global public health challenge. By the end of 2024, approximately 40.8 million people were living with HIV worldwide [<xref ref-type="bibr" rid="ref1">1</xref>]. The widespread use of antiretroviral therapy (ART) has significantly improved life expectancy among people living with HIV [<xref ref-type="bibr" rid="ref2">2</xref>]. However, non-AIDS&#x2013;defining conditions, particularly cardiovascular disease (CVD), are increasingly recognized as critical factors affecting both quality of life and long-term prognosis [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref5">5</xref>]. Epidemiological studies indicate that the risk of CVD among people living with HIV is approximately twice that of the general population, with onset occurring nearly a decade earlier on average [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. The recent REPRIEVE (Randomized Trial to Prevent Vascular Events in HIV) randomized trial further shifted the landscape by demonstrating that proactive statin therapy reduces major adverse cardiovascular events in people living with HIV regardless of traditional risk score thresholds [<xref ref-type="bibr" rid="ref8">8</xref>], highlighting the need for structured prevention approaches in this population.</p><p>In this context, effective CVD management has become a key priority in improving health outcomes among people living with HIV. However, significant interdisciplinary knowledge gaps persist in clinical practice: infectious disease clinicians often lack cardiovascular expertise, while cardiologists may be unfamiliar with ART-related drug interactions. These limitations hinder shared decision-making in areas such as medication selection, risk stratification, and long-term follow-up, potentially leading to suboptimal treatment decisions and adverse clinical outcomes. Traditional medical training has not kept pace with the complexity of HIV-related comorbidities, highlighting the urgent need for innovative tools to support consistent, evidence-based, multidisciplinary care.</p><p>Among these innovative tools, large language models (LLMs) have recently shown great promise in health care applications [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref11">11</xref>]. Built on the transformer architecture, LLMs can process complex language patterns and learn medical knowledge from large-scale textual data [<xref ref-type="bibr" rid="ref12">12</xref>]. Several studies have shown that AI models can generate clinically accurate and interpretable content across diverse specialties [<xref ref-type="bibr" rid="ref13">13</xref>-<xref ref-type="bibr" rid="ref15">15</xref>]. Nonetheless, their utility in managing complex, overlapping comorbidities&#x2014;such as HIV and CVD&#x2014;remains poorly understood. Most prior studies have focused on single disease scenarios [<xref ref-type="bibr" rid="ref16">16</xref>-<xref ref-type="bibr" rid="ref18">18</xref>], and few have examined whether LLMs can synthesize interdisciplinary knowledge, bridge cognitive gaps across specialties, and provide reliable decision support. Notably, different models vary in clinical reasoning depth and update efficiency [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>], which may impact their real-world applicability.</p><p>This study aims to systematically compare the performance of 4 mainstream LLMs&#x2014;DeepSeek-V3, DeepSeek-R1, ChatGPT-4o, and ChatGPT-o4-mini&#x2014;with that of human clinicians in addressing guideline-based CVD management tasks for people living with HIV. We evaluated 4 question categories (basic knowledge, drug management, clinical decision-making, and case analysis) and assessed each response along 4 dimensions: accuracy, completeness, readability, and reliability. By doing so, this study seeks to determine whether LLMs can effectively bridge interdisciplinary knowledge gaps and serve as reliable decision-support tools for managing complex comorbidities in people living with HIV.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>This cross-sectional comparative study systematically evaluated the performance of 4 AI models (DeepSeek-V3, DeepSeek-R1, ChatGPT-4o, and ChatGPT-o4-mini) and 12 clinicians in responding to CVD management questions among people living with HIV. A standardized question set was developed based on authoritative clinical guidelines to ensure consistency across all evaluations (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flowchart of overall study design. CVD: cardiovascular disease; EACS: European AIDS Clinical Society.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89858_fig01.png"/></fig></sec><sec id="s2-2"><title>Question Set Development</title><p>To ensure both the authority and timeliness of the question set, we selected 4 evidence-based guidelines as references. These included the Guidelines for Primary Prevention of Cardiovascular Diseases in China [<xref ref-type="bibr" rid="ref22">22</xref>] and the Chinese Guideline for Diagnosis and Treatment of HIV/AIDS (2024 Edition) [<xref ref-type="bibr" rid="ref23">23</xref>], which reflect local clinical practice, as well as 2 internationally recognized guidelines: the 2023 European AIDS Clinical Society Guidelines [<xref ref-type="bibr" rid="ref24">24</xref>] and Antiretroviral Drugs for Treatment and Prevention of HIV in Adults: 2024 Recommendations of the International Antiviral Society-USA Panel [<xref ref-type="bibr" rid="ref25">25</xref>]. All 4 were developed by leading academic institutions in the fields of CVD and HIV/AIDS, incorporating the most recent high-quality clinical evidence.</p><p>Based on the core content of these guidelines, we adopted a dual-framework strategy that integrates knowledge dimension stratification with clinical needs orientation to screen and classify the questions. Concurrently, we used a 2-round Delphi expert consultation method to systematically refine and optimize the question set. In the first round, 10 experts from multiple disciplines, including cardiology, infectious diseases, and public health, were invited to rate 35 candidate questions across 4 dimensions&#x2014;importance, feasibility, answer clarity, and expression clarity&#x2014;using a 1&#x2010;5 Likert scale, with open-ended suggestions also solicited. Based on the feedback from the first round, we merged, added, and deleted questions to form a standardized set of 25 questions. In the second round of Delphi consultation, the same panel of experts was asked to rerate each question in the revised set (using the same dimensions) and simultaneously evaluate the predefined reference answers in terms of reasonableness and guideline conformity (also using a 1&#x2010;5 Likert scale). Ultimately, questions with mean scores &#x2265;4.0 and coefficients of variation (CVs) &#x2264;0.25 across all dimensions were retained, and textual optimization suggestions from the experts were adopted, resulting in a final structured question set of 25 items.</p><p>The questionnaires used in the 2 rounds of Delphi surveys are detailed in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>This standardized question set follows a progressive framework from basic theory to practical application and then to comprehensive clinical decision-making, covering all key elements of CVD management in people living with HIV. The specific categories are as follows (<xref ref-type="table" rid="table1">Table 1</xref>): the basic theory category focuses on theoretical understanding, assessing respondents&#x2019; mastery of fundamental concepts and standards of CVD management in HIV care; the medication management category addresses 2 common clinical challenges: interactions between ART and cardiovascular drugs, as well as adverse metabolic effects associated with ART; the clinical decision-making category concentrates on formulating diagnostic and treatment plans under specific clinical scenarios, requiring the integration of multidimensional knowledge; and the case analysis category evaluates respondents&#x2019; ability to conduct comprehensive clinical reasoning through complex real-world cases, including risk assessment, treatment planning, and management strategy optimization.</p><p>After the second round of Delphi expert consultation, we finalized reference answers to ensure objective and consistent evaluation. These answers were formulated by prioritizing authoritative guidelines, following a clear hierarchy: Chinese guidelines were prioritized over international guidelines, and guidelines specific to people living with HIV were prioritized over those for the general population, based on the study population&#x2019;s characteristics. We then considered the levels of evidence and clarified points of controversy. This standardized reference then served as the benchmark for evaluation.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Structured question set on CVD<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> management in people living with HIV based on authoritative guidelines<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Category and subcategory</td><td align="left" valign="bottom">Questions</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Basic knowledge</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CVD risk assessment</td><td align="left" valign="top"><list list-type="simple"><list-item><p>1. What cardiovascular risk assessment tools are currently available, and are they applicable to people living with HIV?</p></list-item><list-item><p>2. How is cardiovascular risk stratified into low, medium, and high categories?</p></list-item><list-item><p>3. What are the common cardiovascular risk factors specific to people living with HIV?</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Lifestyle behaviors</td><td align="left" valign="top"><list list-type="simple"><list-item><p>4. What is the recommended upper limit of daily salt intake to reduce CVD risk?</p></list-item><list-item><p>5. What proportion of carbohydrates should be consumed by people living with HIV with elevated blood sugar risk?</p></list-item><list-item><p>6. What level of physical activity is recommended for people living with HIV to reduce CVD risk?</p></list-item><list-item><p>7. How does sleep affect cardiovascular risk, and what are the criteria for healthy sleep?</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Health indicator control</td><td align="left" valign="top"><list list-type="simple"><list-item><p>8. What are the target LDL-C<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> levels for people living with HIV with intermediate, high, and very high cardiovascular risk?</p></list-item><list-item><p>9. What are the recommended blood pressure targets for people living with HIV classified as high cardiovascular risk?</p></list-item><list-item><p>10. What are the optimal blood glucose control standards for people living with HIV to prevent CVD?</p></list-item><list-item><p>11. How frequently should blood pressure and blood glucose be monitored in people living with HIV?</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Drug management</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Drug interactions</td><td align="left" valign="top"><list list-type="simple"><list-item><p>12. Which ART<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> regimens may interact with statins?</p></list-item><list-item><p>13. What are the key considerations when coadministering rilpivirine with other cardiovascular medications, such as calcium channel blockers?</p></list-item><list-item><p>14. What are the risks of drug interactions between tenofovir and diuretics?</p></list-item><list-item><p>15. How do protease inhibitors in ART regimens influence the effectiveness of oral hypoglycemic agents, such as sulfonylureas and metformin?</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Side effects of drugs</td><td align="left" valign="top"><list list-type="simple"><list-item><p>16. Which ART regimens should be avoided in patients with HIV at high risk of CVD?</p></list-item><list-item><p>17. Which ART options may contribute to increased LDL-C levels?</p></list-item><list-item><p>18. Which ART regimens are most likely to cause weight gain?</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Clinical decision-making</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2014;<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top"><list list-type="simple"><list-item><p>19. What are the preferred hypoglycemic treatments for people living with HIV who are at high cardiovascular risk and have type 2 diabetes?</p></list-item><list-item><p>20. What are the stepwise management strategies for people living with HIV with varying blood pressure levels?</p></list-item><list-item><p>21. What are the indications for initiating lipid-lowering therapy in people living with HIV?</p></list-item><list-item><p>22. What lipid-lowering strategies are recommended for patients with different levels of renal dysfunction?</p></list-item><list-item><p>23. What are the causes and management strategies for central adiposity in postmenopausal women receiving dolutegravir-based ART?</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Case analysis</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Case 1</td><td align="left" valign="top"><list list-type="simple"><list-item><p>24. A 45-year-old male with a 10-year history of HIV infection, on ART (EFV<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup>+TDF<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup>+FTC<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup>-based regimen). Current immune status: CD4<sup><xref ref-type="table-fn" rid="table1fn9">i</xref></sup>+580 cells/&#x03BC;L, HIV RNA persistently&#x003C;20 copies/mL (for 3 years). Persistent dyslipidemia (uncontrolled for 3 years): TC<sup><xref ref-type="table-fn" rid="table1fn10">j</xref></sup> 5.2 mmol/L, LDL-C 3.4 mmol/L, HDL-C<sup><xref ref-type="table-fn" rid="table1fn11">k</xref></sup> 1.0 mmol/L, TG<sup><xref ref-type="table-fn" rid="table1fn12">l</xref></sup> 1.8 mmol/L; clinic BP<sup><xref ref-type="table-fn" rid="table1fn13">m</xref></sup> 130/85 mm Hg, home-measured average BP 128/82 mm Hg; FPG<sup><xref ref-type="table-fn" rid="table1fn14">n</xref></sup> 5.8 mmol/L, HbA<sub>1c</sub><sup><xref ref-type="table-fn" rid="table1fn15">o</xref></sup> 5.6%; BMI 26 kg/m<sup>2</sup>, waist circumference 94 cm, eGFR<sup><xref ref-type="table-fn" rid="table1fn16">p</xref></sup> 92 mL/minute/1.73 m<sup>2</sup>. Framingham 10-year risk score 12%, CAC<sup><xref ref-type="table-fn" rid="table1fn17">q</xref></sup> score 50. History of smoking (10 years, 20 pack-years), quit 2 years ago; sedentary office job, fast food&#x2013;based diet.</p></list-item></list><list list-type="alpha-lower"><list-item><p>What are the contributors to dyslipidemia in this patient? Is ART implicated?</p></list-item><list-item><p>Does this patient meet indications for initiating lipid-lowering therapy?</p></list-item><list-item><p>Design a lipid-lowering regimen including drug selection, dosing, and monitoring plan.</p></list-item><list-item><p>Propose nonpharmacologic strategies to improve cardiovascular risk in this patient.</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Case 2</td><td align="left" valign="top"><list list-type="simple"><list-item><p>25. A 58-year-old woman with a 15-year history of HIV infection. Initial ART regimen: LPV/r<sup><xref ref-type="table-fn" rid="table1fn18">r</xref></sup> + TDF/FTC (continued for 13 years). Switched to DTG<sup><xref ref-type="table-fn" rid="table1fn19">s</xref></sup> + 3TC 2 years ago due to mixed hyperlipidemia (peak values: TC 7.2 mmol/L, LDL-C 5.0 mmol/L, TG 4.8 mmol/L). Concurrent conditions: Type 2 diabetes (5-year duration, HbA<sub>1c</sub> 7.5%), currently on metformin 1000 mg bid; hypertension (8-year duration, BP 140-150/85-95 mm Hg), on amlodipine. Latest fasting lipids: TC 6.0 mmol/L, LDL-C 4.0 mmol/L, HDL-C 1.1 mmol/L, TG 2.5 mmol/L; renal function: eGFR 68 mL/minute/1.73 m<sup>2</sup>, UACR<sup><xref ref-type="table-fn" rid="table1fn20">t</xref></sup> 32 mg/g. Cardiovascular risk scores: Framingham 28%, ASCVD<sup><xref ref-type="table-fn" rid="table1fn21">u</xref></sup> 10-year risk 20%, CAC score 450. BMI 28 kg/m<sup>2</sup>, sedentary lifestyle, waist circumference 92 cm.</p></list-item></list><list list-type="alpha-lower"><list-item><p>Based on CVD risk stratification tools (eg, Framingham, ASCVD, and CAC), how should this patient&#x2019;s risk be categorized?</p></list-item><list-item><p>What is the target LDL-C for this patient?</p></list-item><list-item><p>If statin monotherapy fails to achieve target, what are appropriate combination therapy strategies?</p></list-item><list-item><p>Assess the justification for the simplified DTG+3TC regimen considering viral suppression, metabolic effects, and drug interactions.</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>CVD: cardiovascular disease. </p></fn><fn id="table1fn2"><p><sup>b</sup>This table, based on authoritative guidelines, systematically compiles a structured set of questions regarding the management of CVD in people living with HIV.</p></fn><fn id="table1fn3"><p><sup>c</sup>LDL-C: low-density lipoprotein cholesterol.</p></fn><fn id="table1fn4"><p><sup>d</sup>ART: antiretroviral therapy.</p></fn><fn id="table1fn5"><p><sup>e</sup>Not available.</p></fn><fn id="table1fn6"><p><sup>f</sup>EFV: efavirenz. </p></fn><fn id="table1fn7"><p><sup>g</sup>TDF: tenofovir disoproxil fumarate. </p></fn><fn id="table1fn8"><p><sup>h</sup>FTC: emtricitabine. </p></fn><fn id="table1fn9"><p><sup>i</sup>CD4+: cluster of differentiation 4 (a subtype of T-lymphocytes).</p></fn><fn id="table1fn10"><p><sup>j</sup>TC: total cholesterol. </p></fn><fn id="table1fn11"><p><sup>k</sup>HDL-C: high-density lipoprotein cholesterol. </p></fn><fn id="table1fn12"><p><sup>l</sup>TG: triglyceride. </p></fn><fn id="table1fn13"><p><sup>m</sup>BP: blood pressure.</p></fn><fn id="table1fn14"><p><sup>n</sup>FPG: fasting plasma glucose. </p></fn><fn id="table1fn15"><p><sup>o</sup>HbA<sub>1c</sub>: hemoglobin A<sub>1c</sub>. </p></fn><fn id="table1fn16"><p><sup>p</sup>eGFR: estimated glomerular filtration rate. </p></fn><fn id="table1fn17"><p><sup>q</sup>CAC: coronary artery calcium. </p></fn><fn id="table1fn18"><p><sup>r</sup>LPV/r: lopinavir/ritonavir. </p></fn><fn id="table1fn19"><p><sup>s</sup>DTG: dolutegravir. </p></fn><fn id="table1fn20"><p><sup>t</sup>UACR: urinary albumin-to-creatinine ratio.</p></fn><fn id="table1fn21"><p><sup>u</sup>ASCVD: atherosclerotic cardiovascular disease.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-3"><title>AI Model Selection and Response Acquisition</title><p>We selected 4 representative LLMs, namely, DeepSeek-V3, DeepSeek-R1, ChatGPT-4o, and ChatGPT-o4-mini. DeepSeek-V3 is a universal basic model suitable for handling daily tasks; DeepSeek-R1 specializes in complex reasoning and in-depth analytical tasks; ChatGPT-4o features multimodal capabilities and excels in handling routine workflows; and ChatGPT-4o-mini demonstrates notable strengths in science, technology, engineering, and mathematics-related tasks. All models were accessed independently via their respective official public interfaces or clients.</p><p>To ensure consistent and standardized assessment, a unified prompting framework was established for the AI model. The core stipulations embedded in the system prompt included assigning the model the identity of an HIV-specialized clinician, defining 4 categories of research questions (basic knowledge, medication management, clinical decision-making, and case analysis), requiring the model to learn 4 designated clinical guidelines, and setting a fixed evidence hierarchy to resolve inconsistent recommendations across guidelines. The full text of the standardized system prompt is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> for reference.</p><p>Subsequently, the 25 structured questions were input verbatim into each model without any additional instructions attached to individual queries, ensuring that all models received identical information. Each question was submitted only once to avoid potential bias from prior interactions or model memory. Finally, the complete text responses for all 25 questions were collected from each model for subsequent evaluation.</p></sec><sec id="s2-4"><title>Clinician Recruitment and Response Acquisition</title><p>We recruited 12 clinicians from 2 designated hospitals for HIV care in Guangdong Province, China: Shenzhen Third People&#x2019;s Hospital and Dongguan Ninth People&#x2019;s Hospital. The inclusion criteria were as follows: (1) holding a valid medical license and having at least 3 years of clinical experience, (2) being familiar with the diagnosis and treatment of either HIV infection or CVD, and (3) providing informed consent for voluntary participation. Based on their clinical specialties, the clinicians were categorized into 2 groups, with 8 in the infectious diseases group and 4 in the cardiology group. There were no statistically significant differences between the 2 groups in terms of professional title or years of clinical experience, ensuring comparability.</p><p>To balance the response conditions between the AI model and clinicians and control confounding bias, all interviews were conducted in quiet office settings as structured face-to-face verbal question-and-answer sessions with audio recording equipment, which also accommodated clinicians&#x2019; work schedules and facilitated natural responses from participants. Prior to each interview, researchers read a standardized introductory script to participating physicians uniformly (see <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> for details). The script covered the study purpose, response restrictions (no access to any paper or electronic external resources throughout the interview), 4 reference guidelines with a predefined hierarchy of evidence, and classification criteria for the 4 major question categories. Formal questioning only commenced after physicians fully acknowledged comprehension of all rules and granted consent for audio recording. Each interview was scheduled to last 30-60 minutes, while the actual response duration ranged from 20 to 45 minutes (median 30 minutes).</p><p>All audio recordings were transcribed verbatim without subjective alterations to preserve original content. Transcripts were cross-checked by 2 independent researchers to guarantee accuracy and completeness. Any ambiguous or vague statements were verified against the original audio recordings to confirm the intended meaning. To mitigate information bias, transcription and coding were performed under a double-blind protocol; all researchers were blinded to participants&#x2019; identity information and group assignments throughout the entire process.</p></sec><sec id="s2-5"><title>Evaluation Criteria and Implementation</title><p>We constructed a multidimensional evaluation framework encompassing 4 core dimensions: accuracy, completeness, readability, and reliability. The determination of these dimensions was jointly established by multiple experts with clinical and research experience on the research team, through extensive literature searches and reviews followed by several rounds of collective discussion. Specifically, accuracy measures the degree of correctness of medical facts in the response; completeness assesses whether the response covers all key elements involved in the question; readability concerns whether the language expression is clear and the structure is reasonable; and reliability examines whether the response provides reliable guideline-based evidence or a sound clinical reasoning process. These 4 dimensions complement one another and aim to comprehensively characterize the quality of responses from different perspectives. Each dimension was rated on a 4-point scale, where 4 represented the highest performance and 1 the lowest. This system was designed to provide a thorough and detailed evaluation of the responses generated for each question (<xref ref-type="table" rid="table2">Table 2</xref>).</p><p>For the 2 complex case-based questions, which require the integration of multidimensional knowledge and comprehensive clinical reasoning, a single-dimensional score cannot adequately reflect their inherent hierarchical structure. Therefore, based on expert discussion, we developed independent weighted scoring schemes for these 2 case questions (<xref ref-type="table" rid="table3">Table 3</xref>). In these schemes, each subquestion was assigned a different weight according to its clinical relevance and contribution to the overall decision, ensuring that higher-priority elements contributed more substantially to the total score. The specific weight values were also collectively determined by the expert panel through reference to relevant literature and clinical practice consensus. To enhance the clarity and discriminability of the scoring process, we established a standardized rule for handling scores falling between 2 integers&#x2014;when a score lay between adjacent integers, the midpoint of the interval was used as the cutoff value to guide the final scoring decision. This approach facilitated more robust between-group statistical comparisons in subsequent analyses.</p><p>Following the establishment of the evaluation criteria, we convened a panel of 6 experts, comprising 3 public health specialists and 3 clinical physicians. All panel members were independent of the question design and data collection phases. Prior to scoring, the evaluators underwent standardized training to ensure a consistent understanding and application of the scoring criteria. The assessment was conducted using a blinded, independent review protocol. All response texts were anonymized in advance by removing identifying information such as AI model names, clinician names, and institutional affiliations, and each entry was assigned a unique code. Experts then independently evaluated the responses without knowledge of their origin. Upon completion of individual assessments, all score sheets were retrieved and reviewed for interrater consistency.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Definition of evaluation dimensions.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Assessment dimensions and standard description</td><td align="left" valign="bottom">Score setting</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Accuracy</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>The degree to which the answer content aligns with guideline-based evidence, without scientific errors.</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>4=Fully accurate, with no medical errors, and strictly adheres to guideline recommendations.</p></list-item><list-item><p>3=Generally accurate, with only minor expression issues that do not affect the main conclusion.</p></list-item><list-item><p>2=Factually correct overall, but contains evident errors or misleading details.</p></list-item><list-item><p>1=Marginally relevant or contains fundamental scientific inaccuracies.</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Completeness</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>The extent to which the response covers all key points of the question.</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>4=Comprehensive and well-supported response covering all relevant aspects of the question.</p></list-item><list-item><p>3=Covers the main topic, but lacks elaboration or omits some secondary details.</p></list-item><list-item><p>2=Oversimplified answer with missing explanation or &#x2265;2 key elements omitted.</p></list-item><list-item><p>1=Substantial omissions; addresses only a small portion of the question.</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Readability</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>The clarity, coherence, and ease of understanding of the response, facilitating clinical application.</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>4=Well-structured, with appropriate use of bullet points or tables, accurate terminology, and no redundant information.</p></list-item><list-item><p>3=Clear and fluent, though may contain minor segmentation issues or a small amount of irrelevant detail.</p></list-item><list-item><p>2=Logically inconsistent or difficult to follow; requires rereading to grasp.</p></list-item><list-item><p>1=Vague or confusing expression, hindering comprehension.</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Reliability</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>The degree to which the response relies on authoritative sources and aligns with established guidelines.</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>4=Explicitly references high-priority, authoritative guidelines and clearly reflects their content.</p></list-item><list-item><p>3=Consistent with guideline principles but lacks clear attribution to specific sources.</p></list-item><list-item><p>2=Partially aligns with recommendations; relies on less authoritative or unspecified sources.</p></list-item><list-item><p>1=Conflicts with guideline content or lacks credible basis.</p></list-item></list></td></tr></tbody></table></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Weighted scoring framework for case analysis questions based on clinical importance.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Case analysis question</td><td align="left" valign="bottom">Core component</td><td align="left" valign="bottom">Weight (%)</td><td align="left" valign="bottom">Weighted score (score&#x00D7;weight)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Question 24</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>What are the contributors to dyslipidemia in this patient? Is ART<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> implicated?</td><td align="left" valign="top">Etiological analysis</td><td align="left" valign="top">20</td><td align="left" valign="top">0.8</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Does this patient meet indications for initiating lipid-lowering therapy?</td><td align="left" valign="top">Treatment indications</td><td align="left" valign="top">20</td><td align="left" valign="top">0.8</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Design a lipid-lowering regimen including drug selection, dosing, and monitoring plan.</td><td align="left" valign="top">Therapeutic options</td><td align="left" valign="top">40</td><td align="left" valign="top">1.6</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Propose nonpharmacologic strategies to improve cardiovascular risk in this patient.</td><td align="left" valign="top">Nonpharmaceutical intervention</td><td align="left" valign="top">20</td><td align="left" valign="top">0.8</td></tr><tr><td align="left" valign="top" colspan="4">Question 25</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Based on CVD<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> risk stratification tools (eg, Framingham, ASCVD<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup>, and CAC<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup>), how should this patient&#x2019;s risk be categorized?</td><td align="left" valign="top">Risk stratification</td><td align="left" valign="top">20</td><td align="left" valign="top">0.8</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>What is the target LDL-C<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup> for this patient?</td><td align="left" valign="top">Key health indicators</td><td align="left" valign="top">20</td><td align="left" valign="top">0.8</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>If statin monotherapy fails to achieve target, what are appropriate combination therapy strategies?</td><td align="left" valign="top">Combination pharmacotherapy</td><td align="left" valign="top">30</td><td align="left" valign="top">1.2</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assess the justification for the simplified DTG<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup>+3TC<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup> regimen considering viral suppression, metabolic effects, and drug interactions.</td><td align="left" valign="top">Overall clinical decision plan</td><td align="left" valign="top">30</td><td align="left" valign="top">1.2</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>ART: antiretroviral therapy.</p></fn><fn id="table3fn2"><p><sup>b</sup>CVD: cardiovascular disease.</p></fn><fn id="table3fn3"><p><sup>c</sup>ASCVD: atherosclerotic cardiovascular disease.</p></fn><fn id="table3fn4"><p><sup>d</sup>CAC: coronary artery calcium.</p></fn><fn id="table3fn5"><p><sup>e</sup>LDL-C: low-density lipoprotein cholesterol.</p></fn><fn id="table3fn6"><p><sup>f</sup>DTG: dolutegravir.</p></fn><fn id="table3fn7"><p><sup>g</sup>TC: total cholesterol.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-6"><title>Statistical Analysis</title><p>All data analyses were performed using R software (version 4.4.2; R Foundation for Statistical Computing). The primary outcome was the expert-assigned rating scores, which were treated as ordinal categorical variables. To assess data distribution, the Shapiro-Wilk test was used to examine normality, and the Levene test was applied to assess the homogeneity of variances. Descriptive statistics including median and IQR, as well as mean and SD, were used to summarize overall and group-specific performance. Furthermore, to account for the hierarchical structure of the data while respecting the ordinal nature of the outcome variable, we constructed a cumulative link mixed model (CLMM) using the <italic>ordinal</italic> package in R. The model included group membership (AI models vs clinicians) as a fixed effect and random intercepts for both question and rater to capture the clustering of responses. The proportional odds assumption was verified via a likelihood-ratio test comparing models with and without nonproportional odds terms. Post-hoc pairwise comparisons among groups (eg, different AI models) were conducted using the Tukey method for CLMM contrasts, with adjustment for multiple testing.</p><p>To assess the reliability of the expert ratings, interrater agreement among the 6 evaluators was evaluated using the intraclass correlation coefficient (ICC), based on a 2-way random-effects model. Specifically, ICC(2,1) was used to assess the reliability of individual raters, while ICC(2,k) reflected the reliability of the average scores across the panel. An ICC value greater than 0.75 was interpreted as indicating good consistency.</p><p>All statistical tests were 2-sided, and a <italic>P</italic> value of less than .05 was considered statistically significant.</p></sec><sec id="s2-7"><title>Ethical Considerations</title><p>This study received ethics approval from the medical ethics committee of the Medical School of Shenzhen University (approval: PN-202500127). Given that this study solely used anonymous structured interviews with clinicians and a comparative evaluation of responses generated by AI models, no patient samples, clinical medical records, or sensitive personally identifiable information were collected throughout the study. The medical ethics committee approved the waiver of written informed consent. All research procedures were performed in strict accordance with the ethical principles outlined in the Declaration of Helsinki.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Performance Comparison Between AI Models and Clinicians</title><p>This study evaluated the performance of 4 AI models (ChatGPT-4o, ChatGPT-o4-mini, DeepSeek-V3, and DeepSeek-R1) against 12 clinicians (comprising cardiologists and infectious disease clinicians) in answering 25 structured questions related to CVD management in people living with HIV. Each response was independently assessed by 6 experts across 4 core dimensions: accuracy, completeness, readability, and reliability.</p><p>Descriptive analysis showed that the AI group had mean scores ranging from 3.44 to 3.68 across the 4 dimensions, with a median of 4.0 (IQR 3.0-4.0) and CV between 0.145 and 0.178, reflecting relatively high performance consistency. In contrast, the clinician group had mean scores between 1.78 and 2.05, a median of 2.0 (IQR 1.0-2.0), and CV values from 0.428 to 0.473. Across all 4 question types&#x2014;basic knowledge, drug management, clinical decision-making, and case analysis&#x2014;the AI group obtained higher scores than the clinician group. The highest scoring dimension for the AI group was completeness in case analysis questions (mean score 3.8, SD 0.39), while the lowest for clinicians was in drug management (mean score 1.6, SD 0.83). Detailed results are presented in <xref ref-type="table" rid="table4">Table 4</xref> and <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Comparison of performance between the AI model and clinicians across different question categories and evaluation dimensions. The AI model (blue bars) demonstrated significantly higher overall scores compared to clinicians (pink bars) in the majority of categories and dimensions. Error bars represent SD. Asterisks indicate a significant overall effect of the rater group (AI vs clinician) based on a cumulative link mixed model (**<italic>P</italic>&#x003C;.001).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89858_fig02.png"/></fig><p>To formally evaluate differences while accounting for the hierarchical structure of the data (responses nested within questions and raters) and the ordinal nature of the Likert scale, a CLMM was fitted using restricted maximum likelihood. The model demonstrated good fit, with random intercepts for &#x201C;Question&#x201D; (variance=0.0795) and &#x201C;Expert&#x201D; (variance=0.0723), indicating moderate clustering effects. In the fixed effects analysis, the intercept was estimated at &#x2212;4.422 (SE 0.117; <italic>P</italic>&#x003C;.001). The coefficient for the clinician group was &#x2212;5.066 (SE 0.121; <italic>P</italic>&#x003C;.001), corresponding to a significantly lower likelihood of achieving higher scores. Specifically, the odds of the AI group receiving a higher score category compared to the clinician group were 83.6 times greater (odds ratio [OR] 0.012, 95% CI 0.010&#x2010;0.014; <italic>P</italic>&#x003C;.001).</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Comparison of overall distribution between AI models and clinicians.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Group and dimension</td><td align="left" valign="bottom">Mean (SD)</td><td align="left" valign="bottom">Range</td><td align="left" valign="bottom">Median (IQR)</td><td align="left" valign="bottom">CV<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top" colspan="5">AI</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Accuracy</td><td align="left" valign="top">3.63 (0.554)</td><td align="char" char="hyphen" valign="top">1-4</td><td align="left" valign="top">4 (3-4)</td><td align="left" valign="top">0.152</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Completeness</td><td align="left" valign="top">3.68 (0.532)</td><td align="char" char="hyphen" valign="top">1-4</td><td align="left" valign="top">4 (3-4)</td><td align="left" valign="top">0.145</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Readability</td><td align="left" valign="top">3.59 (0.586)</td><td align="char" char="hyphen" valign="top">1-4</td><td align="left" valign="top">4 (3-4)</td><td align="left" valign="top">0.163</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Reliability</td><td align="left" valign="top">3.44 (0.612)</td><td align="char" char="hyphen" valign="top">1-4</td><td align="left" valign="top">4 (3-4)</td><td align="left" valign="top">0.178</td></tr><tr><td align="left" valign="top" colspan="5">Clinicians</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Accuracy</td><td align="left" valign="top">2.00 (0.855)</td><td align="char" char="hyphen" valign="top">1-4</td><td align="left" valign="top">2 (1-3)</td><td align="left" valign="top">0.428</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Completeness</td><td align="left" valign="top">1.87 (0.884)</td><td align="char" char="hyphen" valign="top">1-4</td><td align="left" valign="top">2 (1-2)</td><td align="left" valign="top">0.473</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Readability</td><td align="left" valign="top">2.05 (0.929)</td><td align="char" char="hyphen" valign="top">1-4</td><td align="left" valign="top">2 (1-3)</td><td align="left" valign="top">0.454</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Reliability</td><td align="left" valign="top">1.78 (0.774)</td><td align="char" char="hyphen" valign="top">1-4</td><td align="left" valign="top">2 (1-2)</td><td align="left" valign="top">0.435</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>CV: coefficient of variation.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Performance Comparison Among Different AI Models</title><p>The performance of 4 AI models was further compared across 4 evaluation dimensions and 4 question categories. Descriptive statistics showed that DeepSeek-R1 achieved the highest scores across most evaluation dimensions, particularly in accuracy, with mean scores between 3.79 and 3.87. It performed particularly well in the basic knowledge questions (accuracy: 3.86; completeness: 3.89). Among the comparison group of the other 3 models, the overall performance was largely comparable. ChatGPT-o4-mini exhibited a nuanced advantage in handling complex questions, slightly outperforming others in accuracy (3.73) and readability (3.76) for basic knowledge questions, as well as in completeness (3.83) and readability (3.75) for case analysis questions. This aligns with its ability in managing complex scenarios (<xref ref-type="fig" rid="figure3">Figure 3</xref>).</p><p>To formally evaluate differences among the 4 AI models while accounting for the hierarchical data structure and the ordinal nature of the outcome, a CLMM was fitted, followed by post-hoc pairwise comparisons using the Tukey method. The CLMM revealed a significant overall effect of model type (likelihood-ratio test: <italic>c</italic><sup>2</sup>=24.346; <italic>P</italic>&#x003C;.001). Post-hoc comparisons showed a clear tiered differentiation in performance. DeepSeek-R1 demonstrated a substantial advantage over both ChatGPT-o4-mini and DeepSeek-V3. Specifically, DeepSeek-R1 had significantly higher odds of receiving a higher score compared to ChatGPT-o4-mini (OR 2.47, 95% CI 1.90&#x2010;3.21; <italic>P</italic>&#x003C;.001) and 2.53 times higher than those of DeepSeek-V3 (estimate=0.929; OR 2.53, 95% CI 1.94&#x2010;3.31; <italic>P</italic>&#x003C;.001). In contrast, pairwise comparisons among ChatGPT-4o, ChatGPT-o4-mini, and DeepSeek-V3 revealed no statistically significant differences across any of the evaluated dimensions (adjusted <italic>P</italic> values ranged from .55 to 0.99). For example, the comparison between ChatGPT-4o and ChatGPT-o4-mini yielded an estimate of &#x2212;0.008 (OR 0.992, 95% CI 0.78-1.27; <italic>P</italic>=.99). These results indicate that DeepSeek-R1 occupies a distinct top tier, while the other 3 models perform at a comparable level.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Comparative performance evaluation of 4 AI models. Error bars represent SD, with 4 question types contributing to the mean for each dimension. Significance markers indicate pairwise comparisons based on Tukey-adjusted contrasts from the cumulative link mixed model. **<italic>P</italic>&#x003C;.001<italic>.</italic></p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89858_fig03.png"/></fig></sec><sec id="s3-3"><title>Comparison of Performance Between Different Clinician Groups</title><p>We also compared the performance of cardiologists (n=4) and infectious disease clinicians (n=8) across overall scores, question categories, and evaluation dimensions. Descriptive analysis (consistent with Shapiro-Wilk test results indicating nonnormal distributions: cardiologists <italic>W</italic>=0.837; <italic>P</italic>&#x003C;.001; infectious disease clinicians <italic>W</italic>=0.830; <italic>P</italic>&#x003C;.001) showed comparable central tendencies: the median overall score was 2 (IQR 1.0-3.0) for cardiologists and 2 (IQR 1.0-2.0) for infectious disease clinicians (<xref ref-type="fig" rid="figure4">Figure 4A</xref>). To formally evaluate group differences while accounting for the ordinal nature of the outcome and hierarchical data structure, a CLMM was fitted, followed by Tukey post-hoc pairwise comparisons. The CLMM revealed a significant main effect of clinician group (likelihood-ratio test: <italic>c</italic><sup>2</sup> (1)=19.34; <italic>P</italic>&#x003C;.001).</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Multifaceted comparison of responses among different groups of clinicians. (A) Overall score distribution presented as a boxplot. Mean scores stratified by (B) question category and (C) evaluation dimension. (D) Heat map of scores across categories and dimensions. Group comparisons were performed using a cumulative link mixed model with Tukey-adjusted post-hoc tests (see the Methods section for details).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89858_fig04.png"/></fig><p>Post-hoc comparisons highlighted domain-specific strengths (<xref ref-type="fig" rid="figure4">Figure 4B-D</xref>). In basic knowledge questions, cardiologists scored significantly higher than infectious disease clinicians (median difference 0.25; OR 2.41, 95% CI 1.83&#x2010;3.17; <italic>P</italic>&#x003C;.001), driven by superior performance in CVD risk assessment (accuracy: median 2.26, IQR 1.0-3.0 vs median 1.83, IQR 1.0-3.0; <xref ref-type="fig" rid="figure5">Figure 5</xref>). Conversely, infectious disease clinicians excelled in drug management (median difference &#x2212;0.46; OR 0.41, 95% CI 0.31&#x2010;0.54; <italic>P</italic>&#x003C;.001), particularly in managing ART side effects (accuracy: median 2.23, IQR 1.0-3.0 vs median 1.65, IQR 1.0-2.0; <xref ref-type="fig" rid="figure5">Figure 5</xref>). In clinical decision-making, there was no statistically significant difference between the groups (median difference 0.22; OR 2.09, 95% CI 1.60&#x2010;2.74; adjusted <italic>P</italic>=.16). Similarly, in case analysis, the difference (median difference 0.16; OR 1.71, 95% CI 1.03&#x2010;2.85) also remained nonsignificant after multiple comparison correction (<italic>P</italic><sub>adj</sub>&#x2265;.99 for accuracy).</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Performance comparison of different clinician groups across various problem categories. Data are presented as mean scores for each problem category. Group comparisons were conducted using a cumulative link mixed model with Tukey-adjusted post-hoc tests (see the Methods section for details). CVD: cardiovascular disease.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89858_fig05.png"/></fig><p>By evaluation dimension (<xref ref-type="fig" rid="figure4">Figure 4C and D</xref>), no significant group differences emerged (clinical decision-making: adjusted <italic>P</italic>=.16; case analysis: adjusted <italic>P</italic>&#x2265;.99). For accuracy, cardiologists had marginally higher scores (median 2.07, IQR 1.0-3.0 vs median 1.96, IQR 1.0-2.0), but this was not statistically significant (<italic>P</italic><sub>adj</sub>=.09). Similarly, completeness (median difference=0.02; <italic>P</italic><sub>adj</sub>=.67), readability (median difference=&#x2212;0.09; <italic>P</italic><sub>adj</sub>=.52), and reliability (median difference=0; <italic>P</italic><sub>adj</sub>=.14) showed no meaningful divergence. A heat map (<xref ref-type="fig" rid="figure4">Figure 4D</xref>) further illustrated these specialty-specific patterns: cardiologists consistently scored higher on cardiovascular-related items (eg, CVD risk assessment and health indicator control), whereas infectious disease clinicians excelled in pharmacotherapy-focused tasks (eg, drug interactions and side effect management).</p></sec><sec id="s3-4"><title>Interrater Consistency Analysis</title><p>Interrater agreement was quantified using ICC (<xref ref-type="table" rid="table5">Table 5</xref>). For single-rater reliability (ICC(2,1)), both accuracy (ICC=0.755, 95% CI 0.712-0.794) and completeness (ICC=0.754, 95% CI 0.711-0.793) exceeded the commonly accepted threshold of 0.75 (<italic>P</italic>&#x003C;.001), indicating acceptable consistency among individual expert ratings. In contrast, readability (ICC=0.681) and reliability (ICC=0.713) demonstrated only moderate agreement (<italic>P</italic>&#x003C;.001), suggesting greater variability in individual assessments of these 2 dimensions.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Analysis of interrater consistency (n=6 experts).</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimensions</td><td align="left" valign="bottom">ICC<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>(2,1) (95% CI)</td><td align="left" valign="bottom">ICC(2,k) (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Accuracy</td><td align="left" valign="top">0.755 (0.712-0.794)</td><td align="left" valign="top">0.949 (0.937-0.960)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Completeness</td><td align="left" valign="top">0.754 (0.711-0.793)</td><td align="left" valign="top">0.949 (0.937-0.960)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Readability</td><td align="left" valign="top">0.681 (0.629-0.729)</td><td align="left" valign="top">0.927 (0.910-0.942)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Reliability</td><td align="left" valign="top">0.713 (0.664-0.758)</td><td align="left" valign="top">0.937 (0.923-0.949)</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>ICC: intraclass correlation coefficient.</p></fn></table-wrap-foot></table-wrap><p>To assess the robustness of aggregated expert scoring, average-measures ICC (ICC(2,k)) was further calculated. When ratings were averaged across all 6 experts, excellent consistency was observed across all dimensions: accuracy (ICC=0.949, 95% CI 0.937-0.960), completeness (ICC=0.949, 95% CI 0.937-0.960), readability (ICC=0.927, 95% CI 0.910-0.942), and reliability (ICC=0.937, 95% CI 0.923-0.949; all <italic>P</italic>&#x003C;.001). All values were well above the 0.90 threshold, confirming high reliability of group-based scoring outcomes.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study presents one of the first comprehensive, head-to-head comparisons between LLMs and human clinicians in the context of managing CVD in people living with HIV. By evaluating performance across multiple clinical domains and rating dimensions, our findings consistently demonstrate the superior capability of LLMs in generating accurate, complete, and clinically coherent responses, even within the complex interdisciplinary setting of HIV-CVD comorbidity. This performance advantage was observed across all question types, from basic knowledge to real-world case analysis, and was particularly evident in areas requiring integrative reasoning. Importantly, this advantage emerged despite the AI models receiving only a minimal system prompt specifying their role and the relevant guidelines, rather than being explicitly instructed to mimic a clinician. This suggests that the models&#x2019; pretraining on vast medical corpora inherently equips them with the knowledge and reasoning patterns necessary for complex comorbidity management, underscoring their readiness as decision-support tools.</p><p>This superiority stems from 2 fundamental advantages of LLMs: robust logical reasoning and broad knowledge coverage [<xref ref-type="bibr" rid="ref26">26</xref>-<xref ref-type="bibr" rid="ref28">28</xref>]. Built on advanced algorithmic architectures such as transformers, these models are capable of mining and extracting key features from massive repositories of medical literature, clinical guidelines, consensus statements, and case-based evidence [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. This data-driven integration enables LLMs to conduct multisource, multidimensional information synthesis within milliseconds, allowing them to maintain consistency, accuracy, and completeness in complex clinical scenarios [<xref ref-type="bibr" rid="ref31">31</xref>]&#x2014;qualities that are often compromised in human decision-making due to cognitive biases, fatigue, and fragmented training. These results suggest that, when appropriately deployed, LLMs may serve as a powerful tool for bridging cognitive gaps between specialties and standardizing care quality in settings with limited access to multidisciplinary expertise.</p></sec><sec id="s4-2"><title>Clinical Implications</title><p>It is important to address the clinical relevance of the observed statistical differences. On our 4-point ordinal scale, a score of 4 represents &#x201C;fully accurate, with no medical errors, and strictly adheres to guideline recommendations,&#x201D; while a score of 2 indicates &#x201C;factually correct overall, but contains evident errors or misleading details.&#x201D; The mean AI scores of 3.44&#x2010;3.68 therefore correspond to responses that are near-perfect in guideline adherence, whereas clinician mean scores of 1.78&#x2010;2.05 indicate responses that, on average, contain noticeable inaccuracies or omissions. This qualitative gap has direct implications for patient safety and care quality: even small numerical differences on this scale can translate into clinically meaningful distinctions, such as the difference between recommending a contraindicated drug combination versus a guideline-concordant alternative.</p><p>Furthermore, the extremely low CV in the AI group (CV 0.145&#x2010;0.178) compared with clinicians (CV 0.428&#x2010;0.473) underscores a critical advantage of AI: its ability to deliver uniformly high-quality advice regardless of question complexity or domain. In real-world settings, this consistency can reduce the &#x201C;knowing-doing gap&#x201D; and mitigate variability in clinical decision-making&#x2014;a known driver of disparate patient outcomes. The clinical relevance is most pronounced in the context of complex comorbidities like HIV and CVD, where interdisciplinary knowledge silos are common. An AI model that integrates cardiology and infectious disease guidelines can directly support a clinician in making a safer, more holistic decision than they might achieve alone, particularly in resource-limited settings without easy access to multidisciplinary teams.</p><p>These findings also hint at an underlying complementarity between clinical specialties that merits closer examination, as discussed in the following section.</p></sec><sec id="s4-3"><title>Performance Heterogeneity</title><p>Notable performance differences among the 4 AI models highlight the impact of training strategies and model architecture on clinical reasoning outcomes. DeepSeek-R1 consistently outperformed other models, particularly in tasks requiring factual precision and logical integration, such as basic knowledge and drug safety. This may reflect its specialized training focus on reasoning-intensive tasks and optimized instruction-tuning processes, which likely enhanced its ability to synthesize structured clinical information [<xref ref-type="bibr" rid="ref32">32</xref>]. It is important to note that all models were tested under identical conditions (same prompt and default temperature settings), so the observed differences can be attributed primarily to architectural and training differences rather than prompt engineering.</p><p>In contrast, general-purpose models like ChatGPT-4o and DeepSeek-V3 demonstrated relatively balanced but less domain-specific performance, while ChatGPT-o4-mini showed modest advantages in case analysis, possibly due to its science, technology, engineering, and mathematics&#x2013;oriented pretraining [<xref ref-type="bibr" rid="ref33">33</xref>]. These variations suggest that the clinical utility of LLMs is closely tied to the relevance and quality of their underlying training data. Domain-adaptive pretraining and instruction tuning&#x2014;especially using medical guidelines, real-world case repositories, and multiturn clinical dialogue&#x2014;may significantly improve model performance in complex decision-making scenarios.</p><p>From a translational perspective, these findings underscore the importance of aligning model selection with task characteristics in real-world deployment. For settings requiring high interpretability and decision fidelity, domain-optimized models like DeepSeek-R1 may be preferable. Future efforts should focus on dynamic fine-tuning, multilingual optimization, and integration with local clinical knowledge systems to enhance the safety, generalizability, and cultural adaptability of AI-assisted tools across health care environments.</p></sec><sec id="s4-4"><title>Cross-Specialty Complementarity</title><p>Despite the overall performance gap between AI models and human clinicians, our analysis of subgroup differences revealed meaningful patterns within the clinician group itself [<xref ref-type="bibr" rid="ref34">34</xref>]. Notably, while cardiologists and infectious disease clinicians achieved comparable total scores, their strengths diverged across task types. Cardiologists performed better in cardiovascular risk assessment and complex clinical decision-making, whereas infectious disease clinicians outperformed in drug management tasks, particularly in addressing side effects of drugs. These differences likely reflect each group&#x2019;s clinical focus and training background&#x2014;cardiologists are more experienced with risk stratification algorithms and long-term prognosis planning, while infectious disease clinicians are more attuned to the nuances of ART regimens, metabolic complications, and drug-drug interactions.</p><p>Such findings underscore the existence of &#x201C;knowledge silos.&#x201D; These silos are precisely the gaps that LLMs, with their ability to retrieve and synthesize cross-disciplinary knowledge, can help bridge. This divergence also reinforces the need for more integrated, team-based approaches to care. In real-world practice, the absence of consistent cross-specialty collaboration may lead to fragmented decision-making and missed opportunities for optimized management. From an educational and clinical workflow perspective, integrating LLMs into multidisciplinary team discussions could provide real-time, guideline-aligned suggestions that supplement each specialists&#x2019; blind spots, potentially reducing fragmentation and improving care coordination [<xref ref-type="bibr" rid="ref35">35</xref>].</p><p>Moreover, the observed complementarity between specialties hints at a potential role for LLMs as integrative tools in bridging these knowledge gaps [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. AI-driven decision support systems could offer real-time, cross-domain insights that align with both infectious disease and cardiovascular best practices, thereby supporting more cohesive clinical strategies [<xref ref-type="bibr" rid="ref38">38</xref>]. Future research should explore how such tools can be embedded within multidisciplinary workflows to promote collaborative decision-making and improve outcomes in complex care scenarios.</p></sec><sec id="s4-5"><title>Methodological Contributions</title><p>Beyond the performance findings, this study offers methodological value for future evaluations of AI-assisted clinical decision support. Unlike previous studies that relied on simplified question-answer prompts or single-dimension metrics [<xref ref-type="bibr" rid="ref39">39</xref>], we adopted a standardized, guideline-derived question set, incorporated structured case scenarios, and implemented a multidimensional evaluation framework to capture both the breadth and depth of clinical reasoning. The use of expert-developed reference answers ensured consistency in benchmarking, and the blinded scoring protocol minimized observer bias. Furthermore, we involved experts from both public health and clinical specialties. The interrater agreement, quantified through ICC analysis, demonstrated excellent agreement for average scores across all dimensions (ICC=0.93&#x2010;0.95). These measures ensured that the evaluation process was both reliable and generalizable. This design may serve as a practical assessment paradigm for future studies seeking to objectively measure the clinical utility of LLMs in complex, multidisciplinary contexts.</p></sec><sec id="s4-6"><title>Limitations</title><p>Despite its strengths, this study has several limitations. First, the sample size for clinician participants was relatively small, with only 12 clinicians from 2 HIV-designated hospitals in the same region (Guangdong Province, China). Although we ensured comparable experience levels between specialties and adopted a blinded evaluation approach, the findings may not be generalizable to broader clinical populations or health care systems, and the geographic concentration limits external validity. Second, each AI model was prompted only once per question under standardized conditions, which may not fully reflect their potential performance variability or capacity for interactive refinement&#x2014;especially in real-world deployments where multiturn dialogues are common. Furthermore, the system prompt used (assigning the model the role of a clinician and specifying relevant guidelines) represents just one possible prompting strategy; alternative prompts could lead to different performance outcomes. Third, while our question set was carefully constructed based on authoritative guidelines, it cannot capture the full complexity and uncertainty of clinical practice, such as patient heterogeneity, comorbidities beyond HIV and CVD, or emerging clinical scenarios. Fourth, the weighting scheme for case analysis questions, although determined by expert consensus, introduces inherent subjectivity. Different weighting priorities (eg, emphasizing diagnostic accuracy over treatment planning) could alter the relative rankings of groups. Sensitivity analyses using alternative weighting methods would strengthen the robustness of our findings. Fifth, the evaluation framework, although multidimensional and rigorously implemented, remains inherently subjective, particularly in dimensions such as readability or reliability. Despite acceptable interrater reliability (single-rater ICC=0.68&#x2010;0.76; average-rater ICC=0.93&#x2010;0.95), some degree of rater bias is unavoidable. Finally, the assessment focused on textual response quality and did not evaluate downstream clinical outcomes, patient safety, or acceptability of AI integration in practice.</p></sec><sec id="s4-7"><title>Conclusions</title><p>This study provides the first systematic comparison of LLMs and human clinicians in addressing CVD management for people living with HIV. The results show that AI models achieved significantly higher scores across all evaluation dimensions while also revealing complementary strengths between specialties. These findings highlight the potential of LLMs as decision-support tools that can augment, rather than replace, clinical expertise&#x2014;particularly in complex comorbidity contexts where cross-specialty knowledge integration is essential. The strong domain-specific performance of DeepSeek-R1 further suggests that model selection should consider contextual alignment with real-world clinical needs, beyond size or architecture alone. Future work should prioritize prospective clinical validation and the integration of AI into multidisciplinary workflows, combining human judgment with machine precision to ensure safe and effective deployment.</p></sec></sec></body><back><ack><p>The authors sincerely thank all hospitals, health care professionals, and experts who participated in this study for their valuable contributions. The authors especially thank China Shenzhen Third People&#x2019;s Hospital and Dongguan Ninth People&#x2019;s Hospital for their coordination support and assistance. The authors particularly thank Dr Jiaye Liu, a public health expert, for his methodological support, and Dr Linqin Sun for her assistance during the recruitment of physician volunteers, which significantly enhanced the analytical rigor and reliability of the study results. The authors used the generative AI tool GPT-4.5 (OpenAI) to polish the language and correct grammar in the English manuscript, but this tool was not used for conceptualization, data analysis, result interpretation, or reference generation. All scientific content, research results, and conclusions were independently completed, verified, and approved by the authors. According to the editorial policy of JMIR Publishing Group regarding the use of generative AI, it can be provided upon request.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the National Natural Science Foundation of China (82574171), the project of the Guangdong Basic and Applied Basic Research Foundation (2024A1515012118), and the Medical Science and Technology Foundation of Guangdong Province (A2025250).</p></sec><sec><title>Data Availability</title><p>All data generated or analyzed during this study are included in this manuscript and in the appendices.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ART</term><def><p>antiretroviral therapy</p></def></def-item><def-item><term id="abb2">CLMM</term><def><p>cumulative link mixed model</p></def></def-item><def-item><term id="abb3">CV</term><def><p>coefficient of variation</p></def></def-item><def-item><term id="abb4">CVD</term><def><p>cardiovascular disease</p></def></def-item><def-item><term id="abb5">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">OR</term><def><p>odds ratio</p></def></def-item><def-item><term id="abb8">REPRIEVE</term><def><p>Randomized Trial to Prevent Vascular Events in HIV</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>Data on the size of the HIV epidemic</article-title><source>World Health Organization</source><year>2024</year><access-date>2025-04-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/data/gho/data/themes/hiv-aids/data-on-the-size-of-the-hiv-aids-epidemic">https://www.who.int/data/gho/data/themes/hiv-aids/data-on-the-size-of-the-hiv-aids-epidemic</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="web"><person-group person-group-type="author"><collab>Joint United Nations Programme on HIV/AIDS</collab></person-group><article-title>2017 global AIDS update&#x2014;ending AIDS: progress towards the 90&#x2013;90&#x2013;90 targets</article-title><source>UNAIDS</source><year>2017</year><access-date>2026-06-21</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.unaids.org/en/resources/documents/2017/20170720_Global_AIDS_update_2017">https://www.unaids.org/en/resources/documents/2017/20170720_Global_AIDS_update_2017</ext-link></comment></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>Acquired Immunodeficiency Syndrome Professional Group, Society of Infectious Diseases, Chinese Medical Association; Chinese Center for Disease Control and Prevention</collab></person-group><article-title>Chinese guidelines for the diagnosis and treatment of human immunodeficiency virus infection/acquired immunodeficiency syndrome (2024 edition)</article-title><source>Infect Dis Immun</source><year>2025</year><volume>5</volume><issue>1</issue><fpage>4</fpage><lpage>27</lpage><pub-id pub-id-type="doi">10.1097/ID9.0000000000000152</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dillon</surname><given-names>DG</given-names> </name><name name-style="western"><surname>Gurdasani</surname><given-names>D</given-names> </name><name name-style="western"><surname>Riha</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Association of HIV and ART with cardiometabolic traits in sub-Saharan Africa: a systematic review and meta-analysis</article-title><source>Int J Epidemiol</source><year>2013</year><month>12</month><volume>42</volume><issue>6</issue><fpage>1754</fpage><lpage>1771</lpage><pub-id pub-id-type="doi">10.1093/ije/dyt198</pub-id><pub-id pub-id-type="medline">24415610</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fahme</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Bloomfield</surname><given-names>GS</given-names> </name><name name-style="western"><surname>Peck</surname><given-names>R</given-names> </name></person-group><article-title>Hypertension in HIV-infected adults: novel pathophysiologic mechanisms</article-title><source>Hypertension</source><year>2018</year><month>07</month><volume>72</volume><issue>1</issue><fpage>44</fpage><lpage>55</lpage><pub-id pub-id-type="doi">10.1161/HYPERTENSIONAHA.118.10893</pub-id><pub-id pub-id-type="medline">29776989</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beavers</surname><given-names>C</given-names> </name><name name-style="western"><surname>Pau</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Glidden</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Statin therapy as primary prevention for persons with HIV: a synopsis of recommendations from the U.S. Department of Health and Human Services Antiretroviral Treatment Guidelines Panel</article-title><source>Ann Intern Med</source><year>2025</year><month>06</month><volume>178</volume><issue>6</issue><fpage>847</fpage><lpage>857</lpage><pub-id pub-id-type="doi">10.7326/ANNALS-24-03564</pub-id><pub-id pub-id-type="medline">40418812</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>W</given-names> </name><name name-style="western"><surname>He</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Higher cardiovascular disease risks in people living with HIV: a systematic review and meta-analysis</article-title><source>J Glob Health</source><year>2024</year><month>04</month><day>26</day><volume>14</volume><fpage>04078</fpage><pub-id pub-id-type="doi">10.7189/jogh.14.04078</pub-id><pub-id pub-id-type="medline">38666515</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grinspoon</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Fitch</surname><given-names>KV</given-names> </name><name name-style="western"><surname>Zanni</surname><given-names>MV</given-names> </name><etal/></person-group><article-title>Pitavastatin to prevent cardiovascular disease in HIV infection</article-title><source>N Engl J Med</source><year>2023</year><month>08</month><day>24</day><volume>389</volume><issue>8</issue><fpage>687</fpage><lpage>699</lpage><pub-id pub-id-type="doi">10.1056/NEJMoa2304146</pub-id><pub-id pub-id-type="medline">37486775</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Belkhouribchia</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pen</surname><given-names>JJ</given-names> </name></person-group><article-title>Large language models in clinical nutrition: an overview of its applications, capabilities, limitations, and potential future prospects</article-title><source>Front Nutr</source><year>2025</year><volume>12</volume><fpage>1635682</fpage><pub-id pub-id-type="doi">10.3389/fnut.2025.1635682</pub-id><pub-id pub-id-type="medline">40851903</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jo</surname><given-names>E</given-names> </name><name name-style="western"><surname>Song</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>JH</given-names> </name><etal/></person-group><article-title>Assessing GPT-4&#x2019;s performance in delivering medical advice: comparative analysis with human experts</article-title><source>JMIR Med Educ</source><year>2024</year><month>07</month><day>8</day><volume>10</volume><fpage>e51282</fpage><pub-id pub-id-type="doi">10.2196/51282</pub-id><pub-id pub-id-type="medline">38989848</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Asker</surname><given-names>OF</given-names> </name><name name-style="western"><surname>Recai</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Genc</surname><given-names>YE</given-names> </name><name name-style="western"><surname>Dogan</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Sener</surname><given-names>TE</given-names> </name><name name-style="western"><surname>Sahin</surname><given-names>B</given-names> </name></person-group><article-title>Chatbots in urology: accuracy, calibration, and comprehensibility; is DeepSeek taking over the throne?</article-title><source>BJU Int</source><year>2025</year><month>11</month><volume>136</volume><issue>5</issue><fpage>937</fpage><lpage>945</lpage><pub-id pub-id-type="doi">10.1111/bju.16873</pub-id><pub-id pub-id-type="medline">40741907</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haider</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Prabha</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gomez-Cabello</surname><given-names>CA</given-names> </name><etal/></person-group><article-title>Synthetic patient-physician conversations simulated by large language models: a multi-dimensional evaluation</article-title><source>Sensors (Basel)</source><year>2025</year><month>07</month><day>10</day><volume>25</volume><issue>14</issue><fpage>4305</fpage><pub-id pub-id-type="doi">10.3390/s25144305</pub-id><pub-id pub-id-type="medline">40732431</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Will</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zaretsky</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dowlath</surname><given-names>A</given-names> </name><name name-style="western"><surname>Testa</surname><given-names>P</given-names> </name><name name-style="western"><surname>Feldman</surname><given-names>J</given-names> </name></person-group><article-title>Enhancing the readability of online patient education materials using large language models: cross-sectional study</article-title><source>J Med Internet Res</source><year>2025</year><month>06</month><day>4</day><volume>27</volume><fpage>e69955</fpage><pub-id pub-id-type="doi">10.2196/69955</pub-id><pub-id pub-id-type="medline">40465378</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alamleh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mavedatnia</surname><given-names>D</given-names> </name><name name-style="western"><surname>Francis</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Readability, reliability, and quality analysis of internet-based patient education materials and large language models on Meniere&#x2019;s disease</article-title><source>J Otolaryngol Head Neck Surg</source><year>2025</year><volume>54</volume><fpage>19160216251360651</fpage><pub-id pub-id-type="doi">10.1177/19160216251360651</pub-id><pub-id pub-id-type="medline">40776601</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pal</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wangmo</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bharadia</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Generative AI/LLMs for plain language medical information for patients, caregivers and general public: opportunities, risks and ethics</article-title><source>Patient Prefer Adherence</source><year>2025</year><volume>19</volume><fpage>2227</fpage><lpage>2249</lpage><pub-id pub-id-type="doi">10.2147/PPA.S527922</pub-id><pub-id pub-id-type="medline">40771655</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huo</surname><given-names>B</given-names> </name><name name-style="western"><surname>Boyle</surname><given-names>A</given-names> </name><name name-style="western"><surname>Marfo</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Large language models for chatbot health advice studies</article-title><source>JAMA Netw Open</source><year>2025</year><month>02</month><day>3</day><volume>8</volume><issue>2</issue><fpage>e2457879</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.57879</pub-id><pub-id pub-id-type="medline">39903463</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Evaluating ChatGPT and DeepSeek in postdural puncture headache management: a comparative study with international consensus guidelines</article-title><source>BMC Neurol</source><year>2025</year><volume>25</volume><issue>1</issue><pub-id pub-id-type="doi">10.1186/s12883-025-04280-8</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Assessing the role of large language models between ChatGPT and DeepSeek in asthma education for bilingual individuals: comparative study</article-title><source>JMIR Med Inform</source><year>2025</year><month>08</month><day>13</day><volume>13</volume><pub-id pub-id-type="doi">10.2196/65365</pub-id><pub-id pub-id-type="medline">40802989</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>G&#x00FC;ltekin</surname><given-names>O</given-names> </name><name name-style="western"><surname>Inoue</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yilmaz</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Evaluating DeepResearch and DeepThink in anterior cruciate ligament surgery patient education: ChatGPT&#x2010;4o excels in comprehensiveness, DeepSeek R1 leads in clarity and readability of orthopaedic information</article-title><source>Knee Surg Sports Traumatol Arthrosc</source><year>2025</year><month>08</month><volume>33</volume><issue>8</issue><fpage>3025</fpage><lpage>3031</lpage><pub-id pub-id-type="doi">10.1002/ksa.12711</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eren Korkmaz</surname><given-names>&#x00D6;</given-names> </name><name name-style="western"><surname>A&#x00E7;&#x0131;kal&#x0131;n Ar&#x0131;kan</surname><given-names>B</given-names> </name><name name-style="western"><surname>Say&#x0131;n Kutlu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kaptan Aydo&#x011F;mu&#x015F;</surname><given-names>F</given-names> </name><name name-style="western"><surname>Sezak</surname><given-names>N</given-names> </name></person-group><article-title>Artificial intelligence meets HIV education: comparing three large language models on accuracy, readability, and reliability</article-title><source>Int J STD AIDS</source><year>2026</year><month>02</month><volume>37</volume><issue>2</issue><fpage>112</fpage><lpage>120</lpage><pub-id pub-id-type="doi">10.1177/09564624251372369</pub-id><pub-id pub-id-type="medline">40905356</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jinzhi</surname><given-names>C</given-names> </name><name name-style="western"><surname>Xue</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>Y</given-names> </name></person-group><article-title>Effectiveness of various general large language models in clinical consensus and case analysis in dental implantology: a comparative study</article-title><source>BMC Med Inform Decis Mak</source><year>2025</year><month>03</month><day>26</day><volume>25</volume><issue>1</issue><fpage>147</fpage><pub-id pub-id-type="doi">10.1186/s12911-025-02972-2</pub-id><pub-id pub-id-type="medline">40140812</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>Chinese Society of Cardiology of Chinese Medical Association</collab><collab>Cardiovascular Disease Prevention and Rehabilitation Committee of Chinese Association of Rehabilitation Medicine</collab><collab>Cardiovascular Disease Committee of Chinese Association of Gerontology and Geriatrics</collab><collab>Thrombosis Prevention and Treatment Committee of Chinese Medical Doctor Association</collab></person-group><article-title>Chinese guideline on the primary prevention of cardiovascular diseases</article-title><source>Zhonghua Xin Xue Guan Bing Za Zhi</source><year>2020</year><month>12</month><day>24</day><volume>48</volume><issue>12</issue><fpage>1000</fpage><lpage>1038</lpage><pub-id pub-id-type="doi">10.3760/cma.j.cn112148-20201009-00796</pub-id><pub-id pub-id-type="medline">33355747</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>Acquired Immunodeficiency Syndrome Professional Group of Society of Infectious Diseases, Chinese Medical Association; Chinese Center for Disease Control and Prevention</collab></person-group><article-title>Chinese guideline for diagnosis and treatment of human immunodeficiency virus infection/acquired immunodeficiency syndrome (2024 edition)</article-title><source>Med J Peking Union Med Coll Hosp</source><year>2024</year><volume>15</volume><issue>6</issue><fpage>1261</fpage><lpage>1288</lpage><pub-id pub-id-type="doi">10.12290/xhyxzz.2024-0766</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><article-title>EACS Guidelines Version 13.0</article-title><source>European AIDS Clinical Society</source><year>2024</year><access-date>2026-06-21</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.eacsociety.org/guidelines/eacs-guidelines">https://www.eacsociety.org/guidelines/eacs-guidelines</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gandhi</surname><given-names>RT</given-names> </name><name name-style="western"><surname>Landovitz</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Sax</surname><given-names>PE</given-names> </name><etal/></person-group><article-title>Antiretroviral drugs for treatment and prevention of HIV in adults: 2024 recommendations of the International Antiviral Society-USA Panel</article-title><source>JAMA</source><year>2025</year><month>02</month><day>18</day><volume>333</volume><issue>7</issue><fpage>609</fpage><lpage>628</lpage><pub-id pub-id-type="doi">10.1001/jama.2024.24543</pub-id><pub-id pub-id-type="medline">39616604</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lam</surname><given-names>KT</given-names> </name><name name-style="western"><surname>Yip</surname><given-names>KM</given-names> </name><etal/></person-group><article-title>Comparison of an AI chatbot with a nurse hotline in reducing anxiety and depression levels in the general population: pilot randomized controlled trial</article-title><source>JMIR Hum Factors</source><year>2025</year><month>03</month><day>6</day><volume>12</volume><fpage>e65785</fpage><pub-id pub-id-type="doi">10.2196/65785</pub-id><pub-id pub-id-type="medline">40048637</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pristoupil</surname><given-names>J</given-names> </name><name name-style="western"><surname>Oleaga</surname><given-names>L</given-names> </name><name name-style="western"><surname>Junquero</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Five advanced chatbots solving European Diploma in Radiology (EDiR) text-based questions: differences in performance and consistency</article-title><source>Eur Radiol Exp</source><year>2025</year><month>08</month><day>19</day><volume>9</volume><issue>1</issue><fpage>79</fpage><pub-id pub-id-type="doi">10.1186/s41747-025-00591-0</pub-id><pub-id pub-id-type="medline">40830600</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>D</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>RS</given-names> </name><name name-style="western"><surname>Jomy</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Performance of multimodal artificial intelligence chatbots evaluated on clinical oncology cases</article-title><source>JAMA Netw Open</source><year>2024</year><month>10</month><day>1</day><volume>7</volume><issue>10</issue><fpage>e2437711</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.37711</pub-id><pub-id pub-id-type="medline">39441598</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rahsepar</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Tavakoli</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>GHJ</given-names> </name><name name-style="western"><surname>Hassani</surname><given-names>C</given-names> </name><name name-style="western"><surname>Abtin</surname><given-names>F</given-names> </name><name name-style="western"><surname>Bedayat</surname><given-names>A</given-names> </name></person-group><article-title>How AI responds to common lung cancer questions: ChatGPT vs Google Bard</article-title><source>Radiology</source><year>2023</year><month>06</month><volume>307</volume><issue>5</issue><fpage>e230922</fpage><pub-id pub-id-type="doi">10.1148/radiol.230922</pub-id><pub-id pub-id-type="medline">37310252</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhuang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Accuracy of large language models when answering clinical research questions: systematic review and network meta-analysis</article-title><source>J Med Internet Res</source><year>2025</year><month>04</month><day>30</day><volume>27</volume><fpage>e64486</fpage><pub-id pub-id-type="doi">10.2196/64486</pub-id><pub-id pub-id-type="medline">40305085</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lahat</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sharif</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zoabi</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Assessing generative pretrained transformers (GPT) in clinical decision-making: comparative analysis of GPT-3.5 and GPT-4</article-title><source>J Med Internet Res</source><year>2024</year><month>06</month><day>27</day><volume>26</volume><fpage>e54571</fpage><pub-id pub-id-type="doi">10.2196/54571</pub-id><pub-id pub-id-type="medline">38935937</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ali</surname><given-names>R</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>L</given-names> </name><name name-style="western"><surname>Cui</surname><given-names>H</given-names> </name></person-group><article-title>A comparative study on the use of DeepSeek-R1 and ChatGPT-4.5 in different aspects of plastic surgery</article-title><source>Aesth Plast Surg</source><year>2026</year><month>04</month><volume>50</volume><issue>7</issue><fpage>2776</fpage><lpage>2792</lpage><pub-id pub-id-type="doi">10.1007/s00266-025-05108-z</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ralla</surname><given-names>B</given-names> </name><name name-style="western"><surname>Biernath</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lichy</surname><given-names>I</given-names> </name><etal/></person-group><article-title>How accurate is AI? A critical evaluation of commonly used large language models in responding to patient concerns about incidental kidney tumors</article-title><source>J Clin Med</source><year>2025</year><month>08</month><day>12</day><volume>14</volume><issue>16</issue><fpage>5697</fpage><pub-id pub-id-type="doi">10.3390/jcm14165697</pub-id><pub-id pub-id-type="medline">40869522</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Acharya</surname><given-names>PC</given-names> </name><name name-style="western"><surname>Alba</surname><given-names>R</given-names> </name><name name-style="western"><surname>Krisanapan</surname><given-names>P</given-names> </name><etal/></person-group><article-title>AI-driven patient education in chronic kidney disease: evaluating chatbot responses against clinical guidelines</article-title><source>Diseases</source><year>2024</year><month>08</month><day>16</day><volume>12</volume><issue>8</issue><fpage>185</fpage><pub-id pub-id-type="doi">10.3390/diseases12080185</pub-id><pub-id pub-id-type="medline">39195184</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Metin</surname><given-names>U</given-names> </name><name name-style="western"><surname>Goymen</surname><given-names>M</given-names> </name></person-group><article-title>Information from digital and human sources: a comparison of chatbot and clinician responses to orthodontic questions</article-title><source>Am J Orthod Dentofacial Orthop</source><year>2025</year><month>09</month><volume>168</volume><issue>3</issue><fpage>348</fpage><lpage>357</lpage><pub-id pub-id-type="doi">10.1016/j.ajodo.2025.04.008</pub-id><pub-id pub-id-type="medline">40327024</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alyanak</surname><given-names>B</given-names> </name><name name-style="western"><surname>Dede</surname><given-names>BT</given-names> </name><name name-style="western"><surname>Ba&#x011F;c&#x0131;er</surname><given-names>F</given-names> </name><name name-style="western"><surname>Akaltun</surname><given-names>MS</given-names> </name></person-group><article-title>Parental education in pediatric dysphagia: a comparative analysis of three large language models</article-title><source>J Pediatr Gastroenterol Nutr</source><year>2025</year><month>07</month><volume>81</volume><issue>1</issue><fpage>18</fpage><lpage>26</lpage><pub-id pub-id-type="doi">10.1002/jpn3.70069</pub-id><pub-id pub-id-type="medline">40342137</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Delsoz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hassan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nabavi</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Large language models: pioneering new educational frontiers in childhood myopia</article-title><source>Ophthalmol Ther</source><year>2025</year><month>06</month><volume>14</volume><issue>6</issue><fpage>1281</fpage><lpage>1295</lpage><pub-id pub-id-type="doi">10.1007/s40123-025-01142-x</pub-id><pub-id pub-id-type="medline">40257570</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shiferaw</surname><given-names>MW</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>T</given-names> </name><name name-style="western"><surname>Winter</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mike</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>LN</given-names> </name></person-group><article-title>Assessing the accuracy and quality of artificial intelligence (AI) chatbot-generated responses in making patient-specific drug-therapy and healthcare-related decisions</article-title><source>BMC Med Inform Decis Mak</source><year>2024</year><month>12</month><day>24</day><volume>24</volume><issue>1</issue><fpage>404</fpage><pub-id pub-id-type="doi">10.1186/s12911-024-02824-5</pub-id><pub-id pub-id-type="medline">39719573</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gong</surname><given-names>A</given-names> </name><name name-style="western"><surname>Li</surname><given-names>F</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>R</given-names> </name></person-group><article-title>Evaluating performance of large language models for atrial fibrillation management using different prompting strategies and languages</article-title><source>Sci Rep</source><year>2025</year><volume>15</volume><issue>1</issue><pub-id pub-id-type="doi">10.1038/s41598-025-04309-5</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Construction of a structured question set for CVD management in people living with HIV: A two-round Delphi expert consultation.</p><media xlink:href="jmir_v28i1e89858_app1.pdf" xlink:title="PDF File, 153 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Full AI system prompt.</p><media xlink:href="jmir_v28i1e89858_app2.pdf" xlink:title="PDF File, 47 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Standardized opening script for clinician face-to-face interviews.</p><media xlink:href="jmir_v28i1e89858_app3.pdf" xlink:title="PDF File, 48 KB"/></supplementary-material></app-group></back></article>