<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e93393</article-id><article-id pub-id-type="doi">10.2196/93393</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Performance of 5 Large Language Models in Perioperative Consultation for Pediatric Hypospadias: Cross-Sectional Comparative Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Kang</surname><given-names>Ting</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Yuan</surname><given-names>Chi</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hu</surname><given-names>Xinyu</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Huang</surname><given-names>Wenjiao</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>Department of Pediatric Surgery, West China Hospital of Sichuan University</institution><addr-line>37 Guoxue Xiang, Wuhou District</addr-line><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Brini</surname><given-names>Stefano</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Chen</surname><given-names>Pei-fu</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Banerjee</surname><given-names>Somnath</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Liu</surname><given-names>Zhao</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Wenjiao Huang, MS, Department of Pediatric Surgery, West China Hospital of Sichuan University, 37 Guoxue Xiang, Wuhou District, Chengdu, Sichuan, China, 86 189-8060-6946; <email>bubuhwj@163.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>29</day><month>7</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e93393</elocation-id><history><date date-type="received"><day>12</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>01</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>02</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Ting Kang, Chi Yuan, Xinyu Hu, Wenjiao Huang. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 29.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e93393"/><abstract><sec><title>Background</title><p>Hypospadias is a common congenital malformation requiring surgery. Caregivers face substantial perioperative information needs, and large language models (LLMs) offer a potential health education channel, but their performance in pediatric urology and the relation between citation accuracy and clinical content safety lack systematic evaluation.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate 5 LLMs (ChatGPT-4o, Gemini-2.5-Pro, OpenEvidence, Zhipu Qingyan, and DeepSeek) for pediatric hypospadias perioperative consultation, and to characterize the dimensions clinicians and caregivers prioritize.</p></sec><sec sec-type="methods"><title>Methods</title><p>A noninterventional cross-sectional study was conducted at a tertiary hospital in April 2025. From a 40-item bank, 10 high-priority questions were selected by an independent caregiver screening cohort (N=34, cohort A) and classified into 3 risk levels. Twenty-three pediatric urology experts (6 dimensions) and 36 primary caregivers (cohort B; 4 dimensions) evaluated responses by double-blind forced-ranking (reverse-scored, 5=best). Friedman tests with Kendall <italic>W</italic> assessed overall differences; paired Wilcoxon tests with Bonferroni correction (adjusted &#x03B1;=.005) and rank-biserial <italic>r</italic> with Hodges-Lehmann 95% CIs were used post hoc. Reference authenticity was independently verified by 2 reviewers (XH and WH) using a 5-category scheme (V/PV/F/G/NR [V: Verifiable, PV: Partially Verifiable, F: Fabricated, G: Guideline-Based, Nonspecific, and NR: No References]), with consensus after canonical-source reverification (Cohen &#x03BA;=0.702 preadjudication). An 8-reviewer clinical safety audit (7 senior specialists plus 1 European Association of Urology [EAU]-anchored intermediate-title clinician) applied a 4-level severity scheme (None/Mild/Moderate/Severe).</p></sec><sec sec-type="results"><title>Results</title><p>Models differed significantly (caregiver: <italic>&#x03C7;</italic>&#x00B2;<sub>4</sub>=77.5, <italic>W</italic>=0.538, <italic>P</italic>&#x003C;.001; expert: <italic>&#x03C7;</italic>&#x00B2;<sub>4</sub>=62.2, <italic>W</italic>=0.676, <italic>P</italic>&#x003C;.001). Gemini-2.5-Pro ranked first (expert median 5.0, IQR 3.0&#x2010;5.0; caregiver 4.0, IQR 3.0&#x2010;5.0). DeepSeek ranked second (4.0 both), with superior Empathy versus ChatGPT-4o (<italic>r</italic>=&#x2212;0.343; <italic>P</italic>&#x003C;.001). OpenEvidence scored lowest (2.0 both), despite high citation accuracy. Expert-caregiver agreement was strong (Spearman &#x03C1;=0.89; <italic>P</italic>=.04). Citation accuracy diverged sharply: OpenEvidence was fully verifiable (<italic>V</italic>=100%, <italic>F</italic>=0%), whereas DeepSeek and Zhipu Qingyan showed the highest fabrication (<italic>F</italic> of 33% and 24%, respectively); Gemini-2.5-Pro fabricated none but used nonspecific guideline citations (<italic>G</italic>=85%). The safety audit yielded 78 flags, including 9 Severe-level flags across 5 question&#x2013;model combinations; OpenEvidence carried the largest Severe burden (5 of 9) and the highest severity-weighted score, whereas Gemini-2.5-Pro had the lowest. Bibliographic accuracy and clinical safety were dissociable, and the ranking held under poststratification weighting.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>High citation accuracy does not guarantee clinical safety. In the first dual-perspective evaluation, no model was uniformly best. Gemini-2.5-Pro was most comprehensive but relied on nonspecific guidelines. DeepSeek scored highest on caregiver-rated Empathy, yet it had the highest fabrication rate. OpenEvidence produced the most verifiable citations but carried the heaviest Severe-flag burden. These dimension-level priorities, the dissociation between citation quality and safety, and the portable evaluation framework can inform future pediatric medical&#x2013;artificial intelligence (AI) development. For perioperative use, AI should follow a tiered human-machine collaboration model with mandatory clinician oversight in high-risk scenarios.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>hypospadias</kwd><kwd>perioperative care</kwd><kwd>caregivers</kwd><kwd>artificial intelligence</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Problem</title><p>Hypospadias, a common congenital external genital malformation in male children, exhibits a globally increasing incidence [<xref ref-type="bibr" rid="ref1">1</xref>]. Surgery remains the sole treatment option, yet postoperative complications such as urinary fistula and urethral stricture can directly affect reproductive function and quality of life in adulthood [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]&#xFF0C;placing considerable psychological burdens and information needs on caregivers [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. Current health care resources often fall short of providing timely, personalized guidance, leading caregivers to seek information online&#x2014;where verifying the authenticity of available content presents a persistent challenge [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. The emergence of generative large language models (LLMs) offers a new pathway to mitigate this information asymmetry. Models such as GPT-4 produce medical reasoning and conversational language at near-human fluency [<xref ref-type="bibr" rid="ref8">8</xref>], and earlier work has shown promise in empathetic oncology consultations and accurate ophthalmic advice [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. However, hypospadias sits at the intersection of pediatric complexity and reproductive privacy, where fluent but incorrect advice can be especially harmful [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. Identifying which LLMs can answer caregivers&#x2019; perioperative questions safely and clearly&#x2014;and understanding which evaluation dimensions matter most to clinicians versus families&#x2014; therefore has direct consequences for patient education and equity of access.</p></sec><sec id="s1-2"><title>Review of Relevant Scholarship</title><p>Prior LLM-evaluation work in adjacent settings (eg, bone health queries [<xref ref-type="bibr" rid="ref14">14</xref>], emergency care patient questions [<xref ref-type="bibr" rid="ref11">11</xref>], ophthalmology [<xref ref-type="bibr" rid="ref10">10</xref>], and oncology [<xref ref-type="bibr" rid="ref9">9</xref>]) has typically addressed knowledge-benchmarking or single-perspective accuracy questions. While Faraj et al [<xref ref-type="bibr" rid="ref15">15</xref>] recently evaluated LLM performance on pediatric hypospadias knowledge using a structured Campbell-Walsh multiple-choice questionnaire , their study was restricted to professional-perspective multiple-choice benchmarking. The broader stakeholder needs and evaluation-methodology question addressed here&#x2014;multidimensional, dual-perspective (clinician and caregiver) evaluation of open-ended clinical consultation responses for a sensitive pediatric condition&#x2014;has not been previously characterized.</p><p>While research indicates that performance variations and risks of fabricated or unverifiable citations exist across different models [<xref ref-type="bibr" rid="ref14">14</xref>], and that highly fluent text may mask underlying errors or even dangerous advice [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref13">13</xref>], existing evaluation evidence predominantly relies on English-language contexts. Topics such as pediatric reproductive potential carry distinct cultural sensitivities and taboos within Chinese society. Validating the applicability of Chinese LLMs in this scenario is therefore an urgent priority, yet direct evidence regarding the performance of Chinese models such as DeepSeek in professional medical consultations remains limited [<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>This study differs from prior work in 3 respects. First, instead of single-perspective benchmarking or multiple-choice testing [<xref ref-type="bibr" rid="ref15">15</xref>], we use a dual-perspective (clinician and lay caregiver) evaluation of open-ended consultation responses. Second, the evaluation is stratified by clinical risk level, so sensitive perioperative scenarios are scored separately from routine logistical ones. Third, we run parallel, independent audits of bibliographic citation authenticity and of clinical content safety. Together these 3 components provide a safety-and-utility profile that, to our knowledge, has not previously been reported for pediatric urology artificial intelligence (AI).</p></sec><sec id="s1-3"><title>Hypothesis, Aims, and Objectives</title><p>Building on evaluation framework of Sezgin et al [<xref ref-type="bibr" rid="ref17">17</xref>], this study assesses 5 representative LLMs&#x2014;ChatGPT-4o, Gemini-2.5-Pro, Zhipu Qingyan, DeepSeek, and OpenEvidence&#x2014;in the perioperative consultation context for pediatric hypospadias. The primary aim is to evaluate these models on three components: (1) a double-blind, randomized-label, and forced-ranking comparison from caregiver and expert perspectives; (2) independent citation authenticity verification using a 5-category V/PV/F/G/NR scheme (V: Verifiable, PV: Partially Verifiable, F: Fabricated, G: Guideline-Based, Nonspecific, and NR: No References); and (3) a European Association of Urology (EAU) 2025&#x2013;anchored clinical safety audit. As a secondary aim, ranking robustness is tested by poststratification-weighted sensitivity analyses.</p><p>We prespecified 2 confirmatory hypotheses. First, LLM performance varies across both clinical correctness and patient-centered communication quality, and no single model is uniformly superior on every dimension. Second, stakeholder ratings, citation verifiability, and clinical safety are dissociable: high stakeholder ratings or accurate bibliographic citations do not in themselves guarantee clinical safety. Three additional exploratory aims are (1) whether caregivers, given unrestricted choice, prioritize higher-risk clinical questions; (2) whether model ratings vary systematically across sociodemographic strata or disease severity; and (3) whether expert and caregiver evaluations diverge at the dimension level.</p><p>Each hypothesis maps directly to a design element. The first hypothesis is tested by the double-blind forced-ranking format, which removes central-tendency bias, together with the separate expert and caregiver rubrics that decouple clinical correctness from communication quality. The second hypothesis is tested by reading the stakeholder ranking, the citation verification, and the safety audit side by side rather than collapsing them into a single composite score. The exploratory aims are addressed through stratified subgroup analyses and free-choice question selection, kept separate from the confirmatory tests.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Inclusion and Exclusion</title><sec id="s2-1-1"><title>Expert Evaluators</title><p>The inclusion and exclusion criteria are as follows:</p><list list-type="order"><list-item><p>Inclusion criteria: Practitioners engaged in pediatric urology, pediatric surgery, or related nursing work, holding valid professional practicing licenses.</p></list-item><list-item><p>Exclusion criteria: Unable to complete all evaluation work within the specified period due to heavy clinical workload or other personal reasons.</p></list-item></list></sec><sec id="s2-1-2"><title>Caregiver Participants</title><p>The inclusion and exclusion criteria are as follows:</p><list list-type="order"><list-item><p>Inclusion criteria: Primary caregivers of children receiving perioperative care for hypospadias, able to understand and read standard Chinese medical texts, voluntary participation in this study.</p></list-item><list-item><p>Exclusion criteria: Caregivers with cognitive or mental disorders unable to complete assessments independently, medical staff whose professional knowledge may compromise objective ratings, participants enrolled in other LLM evaluation studies, guardians of children with severe complex comorbidities, and those who refuse consent, withdraw prematurely, or fail to complete all evaluation forms.</p></list-item></list></sec></sec><sec id="s2-2"><title>Participant Characteristics</title><p>The study adopted a 2-cohort design for caregiver recruitment: cohort A (question-bank screening, n=34) and cohort B (LLM response evaluation, n=36), with no overlapping participants between the 2 cohorts. A total of 23 experts were enrolled for professional evaluation. Major demographic characteristics collected for caregivers included age, sex, relationship to the patient, education level, monthly household income, and employment status. Patient (child) characteristics included age and hypospadias type. For the expert panel, demographic and professional characteristics collected included age, clinical experience (years), and professional seniority title. Standardized classifications were applied: monthly household income was categorized using the 2024 per-capita disposable-income data for urban residents from the National Bureau of Statistics of China [<xref ref-type="bibr" rid="ref18">18</xref>] (low &#x2264;3500 RMB/US $&#x2264;488; medium 3501&#x2010;5400 RMB/US $488 to US $752; and high &#x003E;5400 RMB/US $&#x003E;752; 7.18 RMB=US $1 as of April 1, 2025, source: People's Bank of China), and hypospadias types were classified as distal, midshaft, proximal, or severe, which were merged into Distal/Midshaft (Type I and II) and Proximal/Severe (Type III and IV) for subgroup analyses. Expert professional title was categorized into junior, intermediate, and senior titles.</p></sec><sec id="s2-3"><title>Sampling Procedures</title><p>All participants were recruited via convenience sampling at the Department of Pediatric Urology of a tertiary hospital in Chengdu, China. Data collection, including LLM response generation and evaluator recruitment, was completed in April 2025. This manuscript adheres to the Chatbot Assessment Reporting Tool guidelines [<xref ref-type="bibr" rid="ref19">19</xref>] (<xref ref-type="supplementary-material" rid="app14">Checklist 1</xref>) and the APA Journal Article Reporting Standards for Quantitative Research.</p><p>All eligible participants were informed of the study purpose, procedures, and potential risks before enrollment. All participants signed written informed consent voluntarily. No monetary or in-kind compensation was provided to participants. Recruitment of caregivers followed a sequential, nonoverlapping design: cohort A completed the question-bank screening survey; cohort B completed the evaluation. Institutional review board approval and ethical review standards are detailed in Ethical Considerations section.</p></sec><sec id="s2-4"><title>Sample Size, Power, and Precision</title><p>No a priori sample size calculation or power analysis was performed, as the study size was determined by convenience and feasibility within a single specialized tertiary pediatric urology department. Precision of parameter estimates was quantified post hoc using 95% CIs calculated for median scores (exact order-statistic method) and effect-size metrics (rank-biserial correlation and Spearman &#x03C1;). The robustness and generalizability of the achieved sample size against convenience-sampling imbalance were evaluated using a poststratification-weighted sensitivity analysis.</p></sec><sec id="s2-5"><title>Measures and Covariates</title><sec id="s2-5-1"><title>Selection of LLMs and Access Protocol</title><p>Five LLMs (ChatGPT-4o, Gemini-2.5-Pro, OpenEvidence, DeepSeek, and Zhipu Qingyan [ChatGLM and Zhipu AI]) were selected on the basis of general-capability benchmarks from Stanford Human-Centered Artificial Intelligence and iiMedia Research [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>] and on Chinese-language proficiency; OpenEvidence was additionally included as a specialized medical AI model. All 5 were evaluated in their publicly available free-tier web interfaces, without paid subscriptions, plug-ins, or API-level customization. We deliberately tested only free-tier interfaces for four reasons: (1) real-world fidelity&#x2014;caregivers access AI without prompt-engineering expertise or institutional retrieval infrastructure, (2) to establish an unoptimized baseline against which future interventions can be benchmarked, (3) to avoid a model &#x00D7; intervention confound, and (4) to contain evaluator ranking burden. The complete LLM access protocol (URLs, model versions, browser environment, session management, configuration parameters, and the full Chinese/English prompt set) is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. All conclusions are time-stamped to April 6, 2025, and specific to the free-tier interfaces.</p></sec><sec id="s2-5-2"><title>Primary Outcome Measure: Scores From Double-Blind Forced-Ranking Evaluation for LLM Responses</title><p>Reverse scoring was applied (5 points for the best response and 1 point for the worst). Experts evaluated 6 dimensions: Quality, Relevance, Applicability, Source Reliability, Comprehensibility, and Actionability. Caregivers evaluated 4 dimensions: Empathy, Addressing Concerns, Comprehensibility, and Actionability.</p></sec><sec id="s2-5-3"><title>Secondary Outcome Measures</title><sec id="s2-5-3-1"><title>Reference Authenticity</title><p>Reference authenticity was independently verified by 2 trained reviewers (XH and WH) using a standardized search hierarchy (PubMed &#x2192; CrossRef DOI resolver &#x2192; Google Scholar &#x2192; CNKI) and the 5-category V/PV/F/G/NR scheme; definitions in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Cohen &#x03BA; was computed on the classification outcomes; discrepancies were resolved by reapplying the V/PV/F/G/NR framework against canonical sources, and the consensus classification is reported.</p></sec><sec id="s2-5-3-2"><title>Clinical Safety</title><p>An independent clinical safety audit was conducted by 8 reviewers: 7 pediatric urology specialists with senior professional titles (reviewers 1&#x2010;7), supplemented by 1 experienced pediatric urology clinician with an intermediate professional title (reviewer 8) whose evaluation was specifically anchored to the <italic>EAU Guidelines on Paediatric Urology 2025</italic> [<xref ref-type="bibr" rid="ref22">22</xref>], Chapter 3.7 Hypospadias, supplemented by anesthesia and pediatric-surgical literature where the guideline is silent [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>]. All 8 reviewers independently classified each of the 50 blinded responses (5 models &#x00D7; 10 questions) using a 4-level severity scheme: None (Safe), Mild, Moderate, or Severe (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p></sec></sec></sec><sec id="s2-6"><title>Covariates for Stratified Analysis</title><p>Eight covariates were included: caregiver education, family income, employment status, hypospadias type, expert professional seniority, expert age, expert clinical experience, and clinical risk level of consultation questions.</p></sec><sec id="s2-7"><title>Variables for Descriptive Statistics Only</title><p>Caregiver age, relationship to the patient, and patient age were collected for descriptive purposes and not included in formal stratified analysis.</p></sec><sec id="s2-8"><title>Data Collection</title><sec id="s2-8-1"><title>Question Bank and Risk Stratification</title><p>A 40-question bank covering preoperative preparation, surgical treatment, postoperative care, and follow-up was established based on literature and clinical consultation experience [<xref ref-type="bibr" rid="ref26">26</xref>] (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). Two researchers ( TK and CY) independently divided all questions into 3 clinical risk levels (high, medium, and low) according to potential safety hazards caused by incorrect answers. The top 10 most concerned questions selected by cohort A were used for subsequent LLM evaluation.</p></sec><sec id="s2-8-2"><title>LLM Response Generation</title><p>Five LLMs were evaluated via free-tier public web interfaces without paid functions or customized settings. Each model was accessed in Google Chrome with cleared cookies and cache; each question was submitted as an independent single-turn query in a separate conversation window using a standardized parent-to-pediatric-urologist consultation template with an embedded source-request suffix to prevent contextual carryover. Responses were converted to plain text with embedded images, hyperlinks, and markdown formatting removed; self-referential boilerplate phrases (self-introductory framing, self-evaluative meta-commentary, and optional follow-up offers) were uniformly removed across all 5 models to preserve blinding; for the caregiver instrument, bibliographic references were additionally removed as caregivers were not asked to evaluate reference authenticity. Response length was not artificially constrained. The verbatim plain-text responses (50 entries=5 models&#x00D7;10 questions) as presented to evaluators are provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p></sec><sec id="s2-8-3"><title>Evaluation Implementation</title><p>All evaluators received unified pre-evaluation training and calibration sessions before formal assessment. Evaluators completed the independent forced-ranking assessments using randomized, double-blinded digital instruments. We also collected optional open-ended qualitative feedback from caregivers in cohort B after they finished the ranking tasks. These qualitative responses were used to explore caregivers&#x2019; subjective evaluation criteria for LLM outputs. Demographic data were collected via a customized, paper-based questionnaire completed during the baseline recruitment phase.</p></sec><sec id="s2-8-4"><title>Quality of Measurements</title><p>To enhance the quality and reliability of measurements, several procedures were implemented. All evaluators (experts and caregivers) participated in a standardized pre-evaluation briefing that reviewed the operational definition of each evaluation dimension and the rules of the forced-ranking procedure. A calibration session was conducted using sample responses prior to formal data collection to ensure consistent comprehension of the scoring metrics. All data entry was verified independently by 2 members (TK and XH) of the research team to ensure accuracy.</p></sec><sec id="s2-8-5"><title>Instrumentation</title><p>The evaluation instruments comprised three main components: (1) The 40-item candidate question bank, drawn from literature reviews [<xref ref-type="bibr" rid="ref26">26</xref>] and clinical consultation records, spanning 4 perioperative phases (preoperative preparation, surgical treatment, postoperative care, and follow-up) and validated via cohort A screening. (2) The multidimensional evaluation rubric: expert evaluations used a 6-dimension instrument (Quality, Relevance, Applicability, Source Reliability, Comprehensibility, and Actionability) adapted from established large language model evaluation frameworks [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref27">27</xref>] and the Patient Education Materials Assessment Tool [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]; caregiver evaluations used a 4-dimension instrument (Empathy, Addressing Concerns, Comprehensibility, and Actionability) customized for patient communication. Full operational definitions of these dimensions are detailed in <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>. (3) The 5 LLMs themselves, evaluated in their publicly available free-tier web interfaces as detailed in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-8-6"><title>Masking</title><p>A strict double-blind design was implemented throughout the evaluation process. All LLM responses were coded as AI1-AI5, and the label sequence was randomly shuffled for each question to prevent evaluators from identifying specific models. Neither evaluators nor research staff knew the corresponding relationship between codes and actual LLMs during the whole evaluation phase.</p></sec><sec id="s2-8-7"><title>Psychometrics</title><p>Interrater reliability for the reference-authenticity verification task was estimated using Cohen &#x03BA; on the dual-reviewer classification outcomes prior to adjudication. For the clinical safety audit, consistency was assessed across 8 independent reviewers (7 senior specialists and 1 intermediate-title clinician anchored to the <italic>EAU Guidelines on Paediatric Urology 2025</italic> [<xref ref-type="bibr" rid="ref22">22</xref>]), with high interrater convergence defined as agreement on Severe ratings by at least 2 independent reviewers. Psychometric validation of rater scoring was managed through the forced-ranking protocol, which eliminates rater-specific central tendency and scale-usage biases.</p></sec><sec id="s2-8-8"><title>Conditions and Design</title><p>This study used a nonexperimental (observational), noninterventional, and cross-sectional design. No experimental manipulations or randomized assignments of conditions were performed. Five naturally occurring commercial LLM outputs were evaluated across 10 standardized, simulated clinical scenarios. The clinical trial registration was exempted because the study evaluated synthetic, AI-generated text and did not involve any human interventions, prospective assignments, or patient health outcomes.</p></sec><sec id="s2-8-9"><title>Data Diagnostics</title><p>Planned data diagnostics were established prior to analysis. Exclusion criteria post&#x2013;data collection required that any incomplete caregiver or expert evaluations across the 10 questions be excluded from the primary analysis. No data were missing from the completed evaluations, and no statistical outliers were removed. We examined data distribution and statistical assumptions for parametric tests. Since the forced-ranking scores are mutually exclusive and have a fixed total of 15 within each evaluator-question-dimension unit, standard parametric assumptions were not met. For this reason, all analyses were conducted using nonparametric methods. No data transformations or missing data imputations were performed.</p></sec><sec id="s2-8-10"><title>Analytic Strategy</title><p>All statistical analyses were performed in R (version 4.4.0; R Foundation for Statistical Computing). Scores were treated as ordered categorical variables and analyzed with nonparametric methods. The Friedman test (with evaluator ID as the blocking factor) assessed overall differences among the 5 models. To control the family-wise error rate arising from multiple comparisons, significant omnibus results were followed by paired Wilcoxon signed-rank tests with Bonferroni correction across C(5,2)=10 pairwise comparisons (adjusted &#x03B1;=.005). Kendall coefficient of concordance (<italic>W</italic>) was reported as the omnibus effect size, and rank-biserial correlation (<italic>r</italic>) accompanied by Hodges-Lehmann 95% CIs were reported as pairwise effect sizes. Medians were summarized with IQR and 95% CIs computed by exact order-statistic methods (DescTools: MedianCI). Subgroup analyses used the Wilcoxon rank-sum test for 2-level subgroups and both the Kruskal-Wallis omnibus test and the Jonckheere-Terpstra trend test for ordered 3-level subgroups. To assess robustness to convenience-sampling distributional imbalance, a poststratification-weighted sensitivity analysis was conducted using target population distributions from Zhang et al [<xref ref-type="bibr" rid="ref30">30</xref>] for hypospadias severity and education and from Fang et al [<xref ref-type="bibr" rid="ref31">31</xref>] for family income; weighted means and design-based standard errors were computed using the <italic>R</italic> survey package. Practical significance was operationalized by 2 complementary criteria: rank-biserial <italic>r</italic>&#x2265;0.30 (moderate effect, per Cohen [<xref ref-type="bibr" rid="ref32">32</xref>]) and a median rank difference of &#x2265;1 point. No data were missing; no imputation was performed. Where a median&#x2019;s 95% CI collapses to a single point (eg, 5.0&#x2010;5.0), this reflects the bounded 5-point scale with ceiling or floor effects rather than zero variance. Complete omnibus test statistics (<italic>&#x03C7;</italic>&#x00B2;<italic>, df</italic>) for the dimension-, theme-, risk-level-, and subgroup-stratified comparisons are reported in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendices 4</xref> and <xref ref-type="supplementary-material" rid="app7">7</xref><xref ref-type="supplementary-material" rid="app8"/><xref ref-type="supplementary-material" rid="app9"/>-<xref ref-type="supplementary-material" rid="app10">10</xref>.</p></sec></sec><sec id="s2-9"><title>Ethical Considerations</title><p>All procedures complied with the Declaration of Helsinki and were approved by the Biomedical Ethics Review Committee of West China Hospital, Sichuan University (approval number 2025 Review [2457]). All expert and caregiver participants provided written informed consent prior to participation, including consent for anonymized data analysis and publication. This noninterventional, cross-sectional design evaluated AI-generated text rather than assessing human health outcomes; prospective clinical trial registration was not mandated by International Committee of Medical Journal Editors guidelines. Caregiver demographic data were collected anonymously and stored in a password-protected database accessible only to the research team, and no real patient data or identifiable health information was entered into any LLM at any point&#x2014;all queries were based on standardized scenarios from the preconstructed question bank. No monetary, in-kind, or other form of compensation was provided to participants in this study. No images of individual participants, patients, or users are included in the manuscript or any supplementary material.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Participant Flow</title><p>The flow of participants across recruitment, eligibility, allocation, and analysis is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. Caregivers were recruited in 2 sequential phases using independent, nonoverlapping cohorts. For cohort A (question-bank screening), 40 caregivers were assessed for eligibility, of whom 6 (15.0%) were excluded&#x2014;3 did not meet inclusion criteria and 3 declined to participate&#x2014;leaving 34 (85.0%) enrolled who all completed the 40-question survey. For cohort B (LLM response evaluation, independent of cohort A), 40 caregivers were likewise assessed for eligibility, of whom 4 (10.0%) were excluded&#x2014;2 were unable to provide complete data and 2 had children with complex comorbidities&#x2014;leaving 36 (90.0%) enrolled who all completed the double-blind forced-ranking evaluation across the 5 models, 10 questions, and 4 dimensions. The expert panel was recruited via professional referral from the Department of Pediatric Urology and the affiliated nursing department; of 27 invited specialists, 23 (85.2%) completed the full evaluation, while 4 (14.8%) were unable to complete on schedule, primarily due to clinical workload constraints (with no exclusions for conflicts of interest). No data were excluded from analysis from any stream.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Participant flow across the 3 recruitment streams. The diagram tracks cohort A (phase 1 question-bank screening: 40 caregivers assessed; 6 excluded&#x2014;3 did not meet inclusion criteria, 3 declined; 34 enrolled and completed), cohort B (phase II LLM-response evaluation, independent from cohort A: 40 caregivers assessed; 4 excluded&#x2014;2 unable to provide complete data, 2 with complex comorbidities; 36 enrolled and completed), and the expert panel (27 invited; 4 unable to complete on schedule due to clinical workload, with no conflicts of interest; 23 enrolled and completed). EAU: European Association of Urology; LLM: large language model; V/PV/F/G/NR: V: Verifiable, PV: Partially Verifiable, F: Fabricated, G: Guideline-Based, Nonspecific, and NR: No References.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93393_fig01.png"/></fig></sec><sec id="s3-2"><title>Recruitment</title><p>The period of participant recruitment and repeated data collection occurred in April 2025. We recruited 2 independent caregiver cohorts and a panel of clinical experts sequentially. All eligible participants were screened and enrolled following the predefined inclusion and exclusion criteria, and no participants dropped out after formal enrollment.</p></sec><sec id="s3-3"><title>Statistics and Data Analysis</title><sec id="s3-3-1"><title>Baseline Participant Characteristics</title><p>The 23 specialists had a median age of 34 (IQR 29&#x2010;42) years and median clinical experience of 10 (IQR 6&#x2010;20.5) years, distributed across junior (n=9), intermediate (n=7), and senior (n=7) professional titles. The 36 patient-caregiver pairs comprised pediatric patients with a median age of 3.5 (IQR 2&#x2010;6) years and a hypospadias-type distribution that&#x2014;for subgroup analyses&#x2014;was merged into Distal/Midshaft (Type I-II) (26/36, 72.2%) and Proximal/Severe (Type III and IV) (10/36, 27.8%); caregivers were predominantly mothers (28/36, 77.8%), with a median age of 33 (IQR 30&#x2010;36) years, and were distributed across 3 education levels (high school or below [10/36, 27.8%], vocational or junior college [12/36, 33.3%], and undergraduate or above [14/36, 38.9%]) and 3 income strata (low &#x2264;3500 RMB [7/36, 19.4%], medium 3501&#x2010;5400 RMB [10/36, 27.8%], and high &#x003E;5400 RMB [19/36, 52.8%]). Detailed characteristics are summarized in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Baseline characteristics of study participants.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristics</td><td align="left" valign="bottom">Values</td></tr><tr><td align="left" valign="bottom" colspan="2">Patient and primary caregiver characteristics (n=36 pairs)</td></tr></thead><tbody><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Patient characteristics</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age (years), median (IQR)</td><td align="left" valign="top">3.50 (2.00&#x2010;6.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hypospadias type, n (%)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type I (Distal)</td><td align="left" valign="top">15 (41.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type II (Midshaft)</td><td align="left" valign="top">11 (30.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type III (Proximal)</td><td align="left" valign="top">2 (5.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type IV (Severe/Perineal)</td><td align="left" valign="top">8 (22.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Merged: Distal/Midshaft (Types I-II)</td><td align="left" valign="top">26 (72.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Merged: Proximal/Severe (Types III-IV)</td><td align="left" valign="top">10 (27.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Primary caregiver characteristics</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age (years), median (IQR)</td><td align="left" valign="top">33.00 (30.00&#x2010;36.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Relationship to patient, n (%)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mother</td><td align="left" valign="top">28 (77.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Father</td><td align="left" valign="top">8 (22.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Education level, n (%)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>High school or below</td><td align="left" valign="top">10 (27.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Vocational/junior college</td><td align="left" valign="top">12 (33.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Undergraduate or above</td><td align="left" valign="top">14 (38.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Monthly household income (RMB)<sup><xref ref-type="table-fn" rid="table1fn1">a,b</xref></sup>, n (%)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Low (&#x2264;3500)</td><td align="left" valign="top">7 (19.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medium (3501-5400)</td><td align="left" valign="top">10 (27.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>High (&#x003E;5400)</td><td align="left" valign="top">19 (52.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Employment status, n (%)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Unemployed/homemaker</td><td align="left" valign="top">10 (27.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Employed/freelancer</td><td align="left" valign="top">26 (72.2)</td></tr><tr><td align="left" valign="top">Expert panel characteristics (n=23)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age (years), median (IQR)</td><td align="left" valign="top">34.00 (29.00&#x2010;42.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Professional experience (years), median (IQR)</td><td align="left" valign="top">10.00 (6.00&#x2010;20.50)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Professional title, n (%)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Junior</td><td align="left" valign="top">9 (39.1)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Intermediate</td><td align="left" valign="top">7 (30.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Senior</td><td align="left" valign="top">7 (30.4)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>RMB: Chinese Yuan Renminbi.</p></fn><fn id="table1fn2"><p><sup>b</sup>Income categories based on the exchange rate at the time of the study: RMB 7.18=US $1 as of April 1, 2025.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3-2"><title>Question Bank Screening Results</title><p>In the cohort A screening, selection frequencies for the 40 candidate questions ranged from 70.6% (Q13: &#x201C;How long for safe recovery after surgery?&#x201D;) to 38.2% (Q35: &#x201C;What to do if urination difficulty/pain occurs?&#x201D;). The clinical risk distribution was heavily skewed toward low-risk topics, comprising 32 low-risk (80%), 4 medium-risk (10%), and 4 high-risk (10%) questions. However, the top 10 questions carried forward to the LLM-response evaluation demonstrated a markedly enriched proportion of higher-stakes questions: 4 low-risk (40%), 3 medium-risk (30%), and 3 high-risk (30%) (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>).</p></sec><sec id="s3-3-3"><title>Overall Scores for LLM Response Quality</title><p>This study collected scoring data from 23 experts and 36 pediatric caregivers on the 5 LLMs across 10 high-frequency perioperative questions (see <xref ref-type="supplementary-material" rid="app11">Multimedia Appendices 11</xref> and <xref ref-type="supplementary-material" rid="app12">12</xref> for per-question detail). The Friedman test revealed significant differences among the 5 LLMs in both caregiver evaluations (<italic>&#x03C7;</italic>&#x00B2;<sub>4</sub>=77.5, <italic>P</italic>&#x003C;.001; Kendall <italic>W</italic>=0.538) and expert evaluations (<italic>&#x03C7;</italic>&#x00B2;<sub>4</sub>=62.2, <italic>P</italic>&#x003C;.001; Kendall <italic>W</italic>=0.676). Bonferroni-adjusted pairwise comparisons revealed significant differences in most model pairings for caregiver ratings (all <italic>P</italic> adjusted &#x2264;.007). In expert ratings, only 1 comparison failed to reach statistical significance: ChatGPT-4o versus Zhipu Qingyan (<italic>r</italic>=0.455; <italic>P</italic> adjusted=.64); all other pairwise comparisons reached significance (<xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>).</p><p>In overall caregiver rankings (<xref ref-type="fig" rid="figure2">Figure 2A</xref>) and expert ratings (<xref ref-type="fig" rid="figure2">Figure 2B</xref>), Gemini-2.5-Pro ranked first in both groups (expert median 5.0, IQR 3.0&#x2010;5.0, 95% CI 5.0&#x2010;5.0; caregiver median 4.0, IQR 3.0&#x2010;5.0, 95% CI 4.0&#x2010;5.0). DeepSeek ranked second (expert median 4.0, IQR 3.0&#x2010;4.0, 95% CI 4.0&#x2010;4.0; caregiver median 4.0, IQR 3.0&#x2010;4.0, 95% CI 3.0&#x2010;4.0). Zhipu Qingyan ranked third (expert median 3.0, IQR 1.0&#x2010;4.0, 95% CI 2.0&#x2010;3.0; caregiver median 3.0, IQR 2.0&#x2010;4.0, 95% CI 3.0&#x2010;3.0). ChatGPT-4o was in the middle-to-lower range (expert median 3.0, IQR 2.0&#x2010;4.0, 95% CI 3.0&#x2010;3.0; caregiver median 2.0, IQR 2.0&#x2010;4.0, 95% CI 2.0&#x2010;3.0). OpenEvidence received the lowest scores, with expert median 2.0, IQR 1.0&#x2010;2.0, 95% CI 1.0&#x2010;2.0, and caregiver median 2.0, IQR 1.0&#x2010;3.0, 95% CI 1.0&#x2010;2.0.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Distribution of comprehensive scores for response quality of 5 large language models from expert and caregiver perspectives. (A) Expert ratings (n=1380 evaluations per model from 23 experts across 10 questions and 6 dimensions). (B) Caregiver ratings (n=1440 evaluations per model from 36 caregivers across 10 questions and 4 dimensions). Each color band represents the proportion of a given score (1-5) within total evaluations.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93393_fig02.png"/></fig><p>Multidimensional analysis (<xref ref-type="fig" rid="figure3">Figure 3A</xref>, caregiver perspective; and <xref ref-type="fig" rid="figure3">Figure 3B</xref>, expert perspective) revealed that Gemini-2.5-Pro scored consistently high across medical Quality (median 5.0, IQR 4.0&#x2010;5.0; 95% CI 5.0&#x2010;5.0), Applicability (median 5.0, IQR 3.0&#x2010;5.0; 95% CI 4.0&#x2010;5.0), Source Reliability (median 5.0, IQR 3.0&#x2010;5.0; 95% CI 4.0&#x2010;5.0), and Empathy dimensions (median 5.0, IQR 3.0&#x2010;5.0; 95% CI 5.0&#x2010;5.0). DeepSeek maintained stable scores in Quality (median 4.0, IQR 3.0&#x2010;5.0; 95% CI 4.0&#x2010;4.0), Actionability (expert median 4.0, IQR 3.0&#x2010;4.0, 95% CI 3.0&#x2010;4.0; caregiver median 4.0, IQR 3.0&#x2010;4.0, 95% CI 3.0&#x2010;4.0), and Comprehensibility (expert median 4.0, IQR 2.0&#x2010;4.0, 95% CI 3.0&#x2010;4.0; caregiver median 4.0, IQR 3.0&#x2010;4.0, 95% CI 3.0&#x2010;4.0). ChatGPT-4o and Zhipu Qingyan scored at an intermediate level across several dimensions, although caregivers rated ChatGPT-4o lower in Addressing Concerns (median 2.0, IQR 2.0&#x2010;3.0; 95% CI 2.0&#x2010;3.0) and Actionability (median 2.0, IQR 2.0&#x2010;4.0; 95% CI 2.0&#x2010;3.0). OpenEvidence received consistently low scores in Applicability (median 2.0, IQR 1.0&#x2010;3.0; 95% CI 1.0&#x2010;2.0), Source Reliability (median 2.0, IQR 1.0&#x2010;3.0; 95% CI 2.0&#x2010;2.0), and overall Comprehensibility (expert median 1.0, IQR 1.0&#x2010;2.0; 95% CI 1.0&#x2010;2.0).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Score distribution across evaluation dimensions for 5 large language models from experts and caregivers. (A) Caregiver ratings across 4 dimensions (Empathy, Addressing Concerns, Comprehensibility, and Actionability). (B) Expert ratings across 6 dimensions (Quality, Relevance, Applicability, Source Reliability, Comprehensibility, and Actionability). Scoring followed the same forced-ranking reverse-scoring method as in <xref ref-type="fig" rid="figure2">Figure 2</xref> (5=best, 1=worst). Each color band represents the proportion of a given score within total evaluations per dimension. Friedman significance per dimension is annotated; full Bonferroni-corrected pairwise comparisons are shown in <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>. LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93393_fig03.png"/></fig></sec></sec><sec id="s3-4"><title>Model Performance Across Clinical Scenarios and Population Characteristics</title><sec id="s3-4-1"><title>Clinical Consultation Theme Stratification</title><p>A complementary thematic stratification (group A: preoperative and risk-prognosis consultation; group B: postoperative home care guidance; group C: emergency response) is reported in <xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>. The principal pattern is preserved across the 3 themes: Gemini-2.5-Pro dominant, DeepSeek second, and ChatGPT-4o and Zhipu Qingyan intermediate; OpenEvidence lowest.</p></sec><sec id="s3-4-2"><title>Risk-Level Stratification</title><p>Questions were classified into 3 clinical risk levels based on the potential patient safety consequences of incorrect AI-generated information: high risk (Q5, Q6, and Q9), medium risk (Q2, Q7, and Q8), and low risk (Q1, Q3, Q4, and Q10). The Friedman test revealed significant differences among the 5 models at all risk levels (all <italic>P</italic>&#x003C;.001).</p><p>In the caregiver perspective, high-risk questions (<xref ref-type="fig" rid="figure4">Figure 4</xref>B) showed the most pronounced model differences: Gemini-2.5-Pro (median 5.0, IQR 3.0&#x2010;5.0; 95% CI 4.0&#x2010;5.0) and DeepSeek (median 4.0, IQR 3.0&#x2010;4.0; 95% CI 4.0&#x2010;4.0) maintained top-tier performance, while OpenEvidence dropped to median 1.0 (IQR 1.0&#x2010;2.0) (95% CI 1.0&#x2010;1.0). ChatGPT-4o received a low median of 2.0 (IQR 2.0&#x2010;3.0) (95% CI 2.0&#x2010;2.0) in high-risk caregiver questions, significantly lower than DeepSeek (<italic>r</italic>=&#x2212;0.648; <italic>P</italic>&#x003C;.001) and Gemini-2.5-Pro (<italic>r</italic>=&#x2212;0.639; <italic>P</italic>&#x003C;.001).</p><p>In the expert perspective, high-risk questions (<xref ref-type="fig" rid="figure4">Figure 4A</xref>) revealed that Gemini-2.5-Pro (median 5.0, IQR 4.0&#x2010;5.0; 95% CI 5.0-5.0) remained the highest-scoring model, while DeepSeek and ChatGPT-4o converged at median 3.0 (IQR 2.5-4.0, 95% CI 3.0-3.0; and IQR 2.0-4.0, 95% CI 3.0-3.0, respectively). OpenEvidence scored lowest (median 1.0, IQR 1.0&#x2010;2.0; 95% CI 1.0&#x2010;2.0). Notably, the absolute difference in median scores between the highest- and lowest-scoring models increased in the high-risk category (median difference: 4.0; <xref ref-type="fig" rid="figure4">Figure 4A and B</xref>) compared with the low-risk category (median difference: 3.0; <xref ref-type="fig" rid="figure4">Figure 4E and F</xref>) . See <xref ref-type="fig" rid="figure4">Figure 4A-F</xref> and <xref ref-type="supplementary-material" rid="app9">Multimedia Appendix 9</xref> for complete risk-level pairwise comparisons.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Risk-level stratified analysis of 5 large language models&#x2019; response quality, separated by evaluator perspective. Questions were independently classified by 2 research team members into 3 clinical risk levels based on potential patient safety consequences of incorrect information: high risk&#x2014;direct potential harm (Q5: postoperative complications, Q6: complication prevention, and Q9: emergency response to postoperative dysuria), medium risk&#x2014;potential for inappropriate clinical decisions (Q2: long-term outcome, Q7: reproductive or urinary function, and Q8: normal urination assessment), and low risk&#x2014;unlikely to cause direct harm (Q1: success rate, Q3: anesthesia effects, Q4: recovery timeline, and Q10: follow-up schedule). The figure is faceted by perspective (Expert and Caregiver) and within each panel by risk level (High/Medium/Low). Overall, 100% stacked bars display the distribution of scores 1&#x2010;5 for each model under each risk level. Bonferroni-adjusted pairwise comparisons within each cell are reported in <xref ref-type="supplementary-material" rid="app9">Multimedia Appendix 9</xref>. (A) High risk, expert perspective; (B) high risk, caregiver perspective; (C) medium risk, expert perspective; (D) medium risk, caregiver perspective; (E) low risk, expert perspective; and (F) low risk, caregiver perspective.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93393_fig04.png"/></fig></sec><sec id="s3-4-3"><title>Heterogeneity Effects of Population Characteristics</title><p>Stratified analysis examined the influence of socioeconomic characteristics, disease subtype, and expert background on model scoring. Across all 8 stratifications, the principal ranking (Gemini-2.5-Pro&#x003E;DeepSeek&#x003E;Zhipu Qingyan&#x2265;ChatGPT-4o&#x003E;OpenEvidence) proved consistent. On the caregiver side (N=36), no significant differences were observed across education levels (all Kruskal-Wallis <italic>P</italic>&#x003E;.14; all <italic>P</italic> trend &#x003E;.15) or employment status (all <italic>P</italic>&#x003E;.06); for family income, the Kruskal-Wallis tests were nonsignificant for all models (all <italic>P</italic>&#x003E;.05), but the Jonckheere-Terpstra trend tests revealed that OpenEvidence&#x2019;s ratings declined monotonically with increasing income (<italic>P</italic> trend=.033), while Gemini-2.5-Pro showed a marginal rising trend (<italic>P</italic> trend=.094). For disease type, DeepSeek received significantly lower ratings in Proximal or Severe (types III and IV) cases than in distal or midshaft (types I and II) cases (median 3 vs 4; <italic>r</italic>=0.336, <italic>P</italic>=.046). On the expert side (N=23), professional seniority modulated discrimination most strongly: Gemini-2.5-Pro was rated monotonically higher with increasing seniority (Junior median 4/Intermediate median 5/Senior median 5; Kruskal-Wallis <italic>P</italic>=.004; <italic>P</italic> trend &#x003C;.001), OpenEvidence monotonically lower (Junior median 2/Intermediate median 2/Senior median 1; Kruskal-Wallis <italic>P</italic>=.026; <italic>P</italic> trend=.007), and ChatGPT-4o showed a significant trend across seniority (<italic>P</italic> trend=.04) despite a nonsignificant omnibus test. Concordant rising trends were observed for expert age (Gemini-2.5-Pro <italic>P</italic> trend=.024) and clinical experience (Gemini-2.5-Pro <italic>P</italic> trend=.04). Complete stratified analysis tables with descriptive statistics, effect sizes, and 95% CIs are provided in <xref ref-type="supplementary-material" rid="app10">Multimedia Appendix 10</xref>.</p></sec><sec id="s3-4-4"><title>Agreement and Discrepancies Between Experts and Caregivers</title><p>Spearman correlation analysis revealed a strong positive correlation between the overall rankings of the 5 models by both evaluator groups (Spearman &#x03C1;=0.89, 95% CI 0.41&#x2010;1.00; <italic>P</italic>=.04, bootstrap with 2000 resamples). Major discrepancies centered on ChatGPT-4o: the expert group rated its medical Quality at a moderate level (median 3.0, IQR 2.0&#x2010;3.0; 95% CI 2.0&#x2010;3.0), while caregivers scored significantly lower on the Empathy (median 2.5, IQR 2.0&#x2010;4.0; 95% CI 2.0&#x2010;3.0) and Addressing Concerns (median 2.0, IQR 2.0&#x2010;2.0; 95% CI 2.0&#x2010;3.0) dimensions. The caregiver-expert divergence pattern was dimension-specific: expert Source Reliability ratings showed no significant difference between ChatGPT-4o and DeepSeek (<italic>r</italic>=&#x2212;0.127; <italic>P</italic> adjusted=.88), yet the 2 models differed substantially on caregiver-evaluated Empathy (<italic>r</italic>=&#x2212;0.343; <italic>P</italic>&#x003C;.001). Regarding Comprehensibility, DeepSeek and Zhipu Qingyan were well received by caregivers (median 4.0 and 3.0, respectively); in contrast, the specialized medical model OpenEvidence struggled with readability, receiving the lowest Comprehensibility scores from both caregivers (median 2.0) and experts (median 1.0). This strong but imperfect agreement supports the dual-perspective design: the 2 groups converge on overall ranking while diverging on dimension-level priorities.</p><p>Valid qualitative responses were collected from caregivers in cohort B who completed the ranking and provided optional open-ended feedback. Content analysis revealed that caregivers cited &#x201C;level of detail&#x201D; (n=7) and &#x201C;comprehensibility&#x201D; (n=5) as their most frequent ranking criteria, followed by &#x201C;comprehensiveness&#x201D; (n=4), &#x201C;clarity of expression&#x201D; (n=3), and &#x201C;overall impression&#x201D; (n=3). Less frequently mentioned factors included &#x201C;personal preference,&#x201D; &#x201C;patient-centeredness,&#x201D; and &#x201C;intuitiveness of presentation&#x201D; (each n=1). These findings align with the quantitative results, where models scoring higher in structured presentation and readability received more favorable rankings.</p></sec><sec id="s3-4-5"><title>Accuracy of Reference Citations</title><p>Descriptive analysis of the dual-reviewer verification revealed stark contrasts in how the models handled citations (<xref ref-type="table" rid="table2">Table 2</xref>). When resolving discrepancies, the V/PV/F/G/NR scheme was applied conservatively: unretrievable articles were strictly coded as Fabricated (F), and generic guideline mentions lacking specific bibliographic data were classified as Guideline-Based (G).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Reference-authenticity classification by model<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Citations, n</td><td align="left" valign="bottom">V<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>, n (%)</td><td align="left" valign="bottom">PV<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>, n (%)</td><td align="left" valign="bottom">F<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup>, n (%)</td><td align="left" valign="bottom">G<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup>, n (%)</td><td align="left" valign="bottom">NR<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup>, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">OpenEvidence</td><td align="left" valign="top">47</td><td align="left" valign="top">47(100)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top">Gemini-2.5-Pro</td><td align="left" valign="top">40</td><td align="left" valign="top">3(8)</td><td align="left" valign="top">2 (5)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">34 (85)</td><td align="left" valign="top">1 (3)</td></tr><tr><td align="left" valign="top">DeepSeek</td><td align="left" valign="top">39</td><td align="left" valign="top">4 (10)</td><td align="left" valign="top">2 (5)</td><td align="left" valign="top">13 (33)</td><td align="left" valign="top">20 (51)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top">ChatGPT-4o</td><td align="left" valign="top">33</td><td align="left" valign="top">5 (15)</td><td align="left" valign="top">2 (6)</td><td align="left" valign="top">5 (15)</td><td align="left" valign="top">21 (64)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top">Zhipu Qingyan</td><td align="left" valign="top">17</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">1 (6)</td><td align="left" valign="top">4 (24)</td><td align="left" valign="top">10 (59)</td><td align="left" valign="top">2 (12)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Two independent reviewers classified every citation; preadjudication interrater agreement was substantial (raw agreement 80.9%; Cohen &#x03BA;=0.702 across 157 paired records). Discrepancies were resolved by reapplying the V/PV/F/G/NR framework against canonical sources (PubMed &#x2192; CrossRef DOI resolver &#x2192; Google Scholar &#x2192; CNKI); the consensus classification is reported here. Full per-citation classifications are shown in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></fn><fn id="table2fn2"><p><sup>b</sup>V: verifiable.</p></fn><fn id="table2fn3"><p><sup>c</sup>PV: partially verifiable.</p></fn><fn id="table2fn4"><p><sup>d</sup>F: fabricated.</p></fn><fn id="table2fn5"><p><sup>e</sup>G: guideline-based, nonspecific.</p></fn><fn id="table2fn6"><p><sup>f</sup>NR: no references.</p></fn></table-wrap-foot></table-wrap><p>Under these rigorous standards, OpenEvidence emerged as the only model to provide exclusively verifiable citations (<italic>V</italic>=100%, <italic>F</italic>=0%). Gemini-2.5-Pro similarly avoided fabrication (<italic>F</italic>=0%) but instead defaulted heavily to nonspecific guideline referencing (<italic>G</italic>=85%). ChatGPT-4o displayed a more mixed, guideline-leaning profile. In contrast, DeepSeek produced the highest proportion of fabricated references (<italic>F</italic>=33%, representing 13 of the 22 total fabrications), followed by Zhipu Qingyan (<italic>F</italic>=24%). Combined, these 2 models were responsible for 77% (17/22) of all fabricated citations documented in this audit (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>).</p></sec><sec id="s3-4-6"><title>Clinical Safety Audit Findings</title><p>The clinical safety audit identified 78 safety flags across the 5 models, revealing stark disparities in clinical reliability (<xref ref-type="table" rid="table3">Table 3</xref>; <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). All 50 blinded responses were independently evaluated by an 8-member panel. To capture both practical risk and formal compliance, 7 senior pediatric urologists assessed safety through the lens of advanced clinical expertise, while 1 intermediate-title clinician specifically audited the outputs against Chapter 3.7 (Hypospadias) of the updated EAU guidelines [<xref ref-type="bibr" rid="ref22">22</xref>] (<xref ref-type="supplementary-material" rid="app13">Multimedia Appendix 13</xref>).</p><p>Using our predefined severity scale (Severe=3, Moderate=2, Mild=1, and None=0), OpenEvidence registered the highest severity-weighted score (58) and generated the majority of Severe flags (5 of 9). By contrast, Gemini-2.5-Pro maintained the safest profile, accruing a weighted score of just 2 with zero Severe events. Crucially, these safety metrics must be contextualized within the comprehensiveness rankings. A model that generates evasive, low-information responses might naturally avoid safety triggers, yet ultimately provide little actionable guidance to caregivers.</p><p>Across the panel, reviewers issued 9 Severe-level flags tied to 5 unique question-model combinations. Importantly, interrater convergence was strong: 4 of these 5 critical errors were independently identified by multiple reviewers. Qualitative analysis of these flags exposed distinct clinical vulnerabilities among the models.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Clinical safety audit summary<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Severe</td><td align="left" valign="bottom">Moderate</td><td align="left" valign="bottom">Mild</td><td align="left" valign="bottom">Total flags</td><td align="left" valign="bottom">Severity-weighted score</td></tr></thead><tbody><tr><td align="left" valign="top">Zhipu Qingyan</td><td align="left" valign="top">2</td><td align="left" valign="top">9</td><td align="left" valign="top">9</td><td align="left" valign="top">20</td><td align="left" valign="top">33</td></tr><tr><td align="left" valign="top">Gemini-2.5-Pro</td><td align="left" valign="top">0</td><td align="left" valign="top">0</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td></tr><tr><td align="left" valign="top">OpenEvidence</td><td align="left" valign="top">5</td><td align="left" valign="top">9</td><td align="left" valign="top">25</td><td align="left" valign="top">39</td><td align="left" valign="top">58</td></tr><tr><td align="left" valign="top">DeepSeek</td><td align="left" valign="top">0</td><td align="left" valign="top">4</td><td align="left" valign="top">6</td><td align="left" valign="top">10</td><td align="left" valign="top">14</td></tr><tr><td align="left" valign="top">ChatGPT-4o</td><td align="left" valign="top">2</td><td align="left" valign="top">3</td><td align="left" valign="top">2</td><td align="left" valign="top">7</td><td align="left" valign="top">14</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>The audit was conducted by 8 independent reviewers evaluating 50 blinded responses (10 questions&#x00D7;5 models). The severity-weighted score is calculated using the following grading scheme: Severe=3, Moderate=2, Mild=1, and None (Safe)=0.</p></fn></table-wrap-foot></table-wrap><p>OpenEvidence alone accounted for the majority of Severe events (5 of 9), stemming from 3 distinct clinical misjudgments. First, it inappropriately recommended routine preoperative androgen stimulation for distal hypospadias (Q6). This violates the EAU 2025 guidelines (&#x00A7;3.7.5.2), which strictly reserve hormonal therapy for specific proximal cases or micropenis. Second, rather than correctly triaging postoperative dysuria as a potential surgical emergency&#x2014;such as urinary retention or catheter obstruction&#x2014;the model provided an abstract academic differential diagnosis (Q9). Finally, it dangerously understated the timeline for full clinical recovery (Q4). Reviewers also frequently flagged OpenEvidence for moderate errors and epidemiological misrepresentations. For instance, it framed clinic-based uroflowmetry as a home-monitoring task (Q8), exaggerated fistula rates up to 70% (Q5), and presented high reoperation rates (51.8%) from complex redo surgeries without necessary case-mix caveats (Q1).</p><p>Other models similarly generated clinically hazardous advice. ChatGPT-4o produced an internally contradictory response advising an excessively prolonged catheter indwelling time of approximately 24 weeks (Q4)&#x2014;a stark deviation from EAU 2025 &#x00A7;3.7.5.8, which notes durations ranging only from zero days to a few weeks. Zhipu Qingyan&#x2019;s 2 Severe flags resulted from recommending routine retrograde urethrography at 3 and 6 months postoperatively (Q10). This invasive, radiation-exposing study is inappropriate for routine pediatric surveillance and directly conflicts with the EAU recommendation for noninvasive uroflowmetry (&#x00A7;3.7.7.1). Conversely, Gemini-2.5-Pro generated no Severe events and produced zero fabricated citations (<italic>F</italic>=0%), although it relied heavily on nonspecific, Guideline-Based citations (G=85%).</p><p>Ultimately, this stark dissociation between citation authenticity and clinical safety stands as a principal finding of our study. The most bibliographically accurate model&#x2014;OpenEvidence (V=100%)&#x2014;paradoxically accumulated the highest clinical safety burden. In contrast, DeepSeek, which recorded the highest citation fabrication rate (<italic>F</italic>=33% and, alongside Zhipu Qingyan, accounted for 77% of all fake citations), generated zero Severe safety events. This provides direct evidence that high citation accuracy does not guarantee clinical safety; they are fundamentally dissociable dimensions that necessitate separate evaluation (for the complete EAU 2025&#x2013;anchored clinical safety audit and specific chapter-section anchors, see <xref ref-type="supplementary-material" rid="app3">Multimedia Appendices 3</xref> and <xref ref-type="supplementary-material" rid="app13">13</xref>).</p></sec><sec id="s3-4-7"><title>Poststratification-Weighted Sensitivity Analysis</title><p>To ensure that findings were not artifacts of convenience sampling, poststratification weighting was applied, adjusting for hypospadias severity, caregiver education, and family income based on published national pediatric distributions [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Weights were normalized to the sample size (N=36), and weighted means with design-based standard errors were computed using the <italic>R</italic> survey package. This sensitivity analysis yielded rank orders identical to the unweighted primary analysis (Spearman &#x03C1;=1.00) (<xref ref-type="table" rid="table4">Table 4</xref>):</p><p>Although confidence intervals between adjacent ranks showed limited overlap due to design-based variance inflation, separation widened substantially for nonadjacent pairs, providing strong evidence that the principal model rankings are robust and generalizable.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Poststratification-weighted ranking of model performance in sensitivity analysis.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Rank</td><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Weighted mean</td><td align="left" valign="bottom">Design-based 95% CI</td></tr></thead><tbody><tr><td align="left" valign="top">1</td><td align="left" valign="top">Gemini-2.5-Pro</td><td align="left" valign="top">3.75</td><td align="left" valign="top">3.39&#x2010;4.11</td></tr><tr><td align="left" valign="top">2</td><td align="left" valign="top">DeepSeek</td><td align="left" valign="top">3.43</td><td align="left" valign="top">3.18&#x2010;3.68</td></tr><tr><td align="left" valign="top">3</td><td align="left" valign="top">Zhipu Qingyan</td><td align="left" valign="top">3.07</td><td align="left" valign="top">2.87&#x2010;3.26</td></tr><tr><td align="left" valign="top">4</td><td align="left" valign="top">ChatGPT-4o</td><td align="left" valign="top">2.59</td><td align="left" valign="top">2.42&#x2010;2.76</td></tr><tr><td align="left" valign="top">5</td><td align="left" valign="top">OpenEvidence</td><td align="left" valign="top">2.16</td><td align="left" valign="top">1.90&#x2010;2.43</td></tr></tbody></table></table-wrap></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Support of Original Hypotheses</title><p>This study evaluated 5 mainstream LLMs&#x2014;Gemini-2.5-Pro, DeepSeek, ChatGPT-4o, Zhipu Qingyan, and OpenEvidence&#x2014;for perioperative consultation in pediatric hypospadias. The first confirmatory hypothesis&#x2014;that LLM performance would vary across clinical correctness and patient-centered communication, with no model proving uniformly superior&#x2014;was supported. Gemini-2.5-Pro produced the most comprehensive answers in both expert and caregiver evaluations; DeepSeek ranked second, with caregivers giving particularly high marks on Empathy; and OpenEvidence finished last from both perspectives despite producing the highest verifiable citation rate. No model led on every dimension. The principal ranking remained stable across 8 sociodemographic strata, across the 3 clinical risk levels, and across the poststratification-weighted sensitivity analysis (Spearman &#x03C1;=1.00 between weighted and unweighted rankings), arguing against demographic or clinical confounding as drivers of the observed pattern.</p><p>The second confirmatory hypothesis&#x2014;that stakeholder ratings, citation verifiability, and clinical safety are dissociable dimensions of model behavior&#x2014;was also supported. The 3 pillars yielded inconsistent rankings of the same models. OpenEvidence had the highest citation accuracy (<italic>V</italic>=100%) and yet accumulated 5 of 9 (56%) panel-wide Severe events, showing that accurate citations do not, on their own, guarantee clinically safe content; whether they are necessary cannot be answered from cross-sectional data of this design. DeepSeek illustrated the inverse: a citation fabrication rate of 33% co-occurred with zero Severe events. Gemini-2.5-Pro, also free of fabricated citations (<italic>F</italic>=0%), likewise recorded none. Fabricated citations and unsafe clinical content therefore appear to be distinct failure modes that require different mitigation strategies.</p><p>The 3 exploratory aims, reported with caution given uncontrolled family-wise error rates, each yielded a substantive finding. When caregivers were given unrestricted choice, they disproportionately selected higher-risk clinical questions: High- and medium-risk items accounted for 60% of the final top 10 despite contributing only 20% of the original pool, suggesting that caregivers naturally weight high-stakes clinical information above routine logistical queries, consistent with the heightened concern documented among caregivers of children with serious conditions [<xref ref-type="bibr" rid="ref5">5</xref>], as well as their active information-seeking behaviors [<xref ref-type="bibr" rid="ref6">6</xref>] and raising the bar for accuracy and safety in any AI model used in real-world patient education. Subgroup analysis showed a monotonic decline in OpenEvidence ratings as caregiver income rose (<italic>P</italic> trend =.033), with a marginal trend in the opposite direction for Gemini-2.5-Pro; higher-income caregivers, who may have greater health literacy [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>] or stronger expectations about delivery style, appear less tolerant of OpenEvidence&#x2019;s academic phrasing. On the Empathy dimension, caregivers rated DeepSeek and Zhipu Qingyan higher than ChatGPT-4o, although the underlying mechanisms&#x2014;linguistic familiarity, cultural framing, or affective resonance&#x2014;were not measured directly here and remain hypotheses for future work.</p></sec><sec id="s4-2"><title>Similarity of Results</title><p>Most prior work on medical AI has focused on model accuracy and treated accuracy as a proxy for usefulness [<xref ref-type="bibr" rid="ref10">10</xref>]. Our sociodemographic data complicate that picture. The monotonic decline in OpenEvidence ratings with rising income, alongside the marginal rising trend for Gemini-2.5-Pro in the same group, mirrors earlier observations [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>] that specialized information presented without lay-audience adaptation can hinder patient decision-making [<xref ref-type="bibr" rid="ref35">35</xref>]. The caregiver qualitative comments point in the same direction: level of detail and comprehensibility were the 2 most frequent reasons for ranking. Deploying AI tools without sufficient attention to comprehensibility risks widening, rather than narrowing, disparities in access to understandable medical information [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]&#x2014;a health-equity issue that must be explicitly addressed in all AI deployment plans.</p><p>On the affective dimensions, the locally adapted models DeepSeek and Zhipu Qingyan were rated higher on Empathy by caregivers than ChatGPT-4o. This is consistent with Masoud et al [<xref ref-type="bibr" rid="ref37">37</xref>], who showed that language models tuned to specific linguistic and cultural contexts more closely reflect local social norms. ChatGPT-4o&#x2019;s pattern was characteristic: experts gave it moderate Quality scores, but caregivers rated it poorly on Empathy and Addressing Concerns. Mesko and Topol [<xref ref-type="bibr" rid="ref38">38</xref>] have argued that the strict safety alignment training of general-purpose models can push them toward formulaic disclaimers that read as cold; our caregiver ratings are compatible with that account. The trade-off is real, however: DeepSeek&#x2019;s higher Empathy was paired with the highest citation-fabrication rate; so the Chinese-adapted models in our panel have not yet reconciled empathy with citation reliability.</p><p>The dissociation between citation accuracy and clinical safety extends earlier concerns. As demonstrated in health information overload research, highly technical evidence presented without clinical context can function as &#x201C;information noise&#x201D; that impairs patient comprehension, and Athaluri et al [<xref ref-type="bibr" rid="ref12">12</xref>] and Alkaissi and McFarlane [<xref ref-type="bibr" rid="ref13">13</xref>] documented the misinformation risk posed by fabricated citations that are superficially credible. Our V/PV/F/G/NR consensus classification adds resolution: DeepSeek and Zhipu Qingyan together account for the majority of fabricated citations, while Gemini-2.5-Pro&#x2019;s guideline-leaning strategy prevents fabrication at the cost of bibliographic granularity. Within the disease-severity dimension, DeepSeek&#x2019;s statistically significant performance drop in Proximal or Severe (types III and IV) cases (<italic>r</italic>=0.336, moderate effect) suggests that probabilistic models still have trouble matching the case-specific judgment human experts apply in complex hypospadias surgery&#x2014;consistent with Topol&#x2019;s [<xref ref-type="bibr" rid="ref39">39</xref>] view (and GPT-4 analysis by Lee et al [<xref ref-type="bibr" rid="ref40">40</xref>]) that human and machine intelligence work best in combination, not substitution. A practical translation is the tiered human-machine workflow proposed by Liyanage et al [<xref ref-type="bibr" rid="ref41">41</xref>] and Krajcer [<xref ref-type="bibr" rid="ref42">42</xref>]: validated models handle low-risk educational queries, while severe deformities, complication management, and complex surgical planning stay with human specialists. This is in line with LLM evaluation work in urolithiasis [<xref ref-type="bibr" rid="ref43">43</xref>] and cornea care [<xref ref-type="bibr" rid="ref44">44</xref>], where comprehensibility and triage gaps were similarly identified.</p></sec><sec id="s4-3"><title>Interpretation</title><p>Several sources of bias and threats to internal validity should be kept in mind. To counter the central tendency bias inherent in traditional absolute rating scales [<xref ref-type="bibr" rid="ref45">45</xref>], we used a double-blind forced-ranking design with per-question randomized labels. This design forced evaluators to make distinct comparative judgments, effectively curbing &#x201C;satisficing&#x201D;&#x2014;the tendency to default to neutral scores when cognitive demands are high [<xref ref-type="bibr" rid="ref46">46</xref>]. As a result, the internal validity of our between-model comparisons was strengthened. Sampling was nevertheless by convenience at a single tertiary center in Chengdu, China, and the expert and caregiver groups likely underrepresent primary care, rural, and other cultural settings; with mothers making up 28 of 36 caregivers (77.8%), the Empathy and communication-quality findings cannot be assumed to extend to fathers or to other family structures.</p><p>Evaluator seniority represents a key potential source of assessment bias. Stratified analysis revealed that senior specialists rated Gemini-2.5-Pro monotonically higher and OpenEvidence monotonically lower as seniority increased, likely reflecting a higher sensitivity to citation-fabrication risks or stricter clinical standards. This aligns with recent evidence from Faraj et al [<xref ref-type="bibr" rid="ref15">15</xref>], who found that senior, certified pediatric urologists exhibit significantly sharper clinical reasoning in hypospadiology assessments than junior peers. Consequently, accounting for evaluator seniority is essential, as long-term clinical maturity dictates a more rigorous, risk-aware standard when auditing AI-generated counseling content.</p><p>Regarding imprecision of measurement protocols, constructs such as eHealth literacy, cultural identity, linguistic preference, and state anxiety were not measured using validated psychometric instruments (eg, the eHealth Literacy Scale [eHEALS] [<xref ref-type="bibr" rid="ref36">36</xref>] and the State-Trait Anxiety Inventory [STAI] [<xref ref-type="bibr" rid="ref47">47</xref>]).</p><p>The overall number of tests and overlap among tests merit consideration. The principal ranking proved robust across 8 sociodemographic strata, 3 clinical risk levels, and poststratification weighting, providing strong evidence that the findings are not artifacts of multiple testing. However, subgroup analyses and exploratory free-choice question findings should be interpreted as hypothesis-generating rather than confirmatory. The adequacy of sample sizes and sampling validity is supported by the same robustness checks (ie, stratification and weighting), but the single-site, single-disease design still limits generalizability beyond this cohort.</p></sec><sec id="s4-4"><title>Generalizability</title><p>Several contextual factors limit how far these results travel. All 5 models were tested at one moment in time (April 6, 2025) on free-tier web interfaces with no exposed version identifiers, so the ranking values are explicitly time-anchored [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref48">48</xref>]; the evaluation method itself, and the clinician- and caregiver-prioritized dimensions identified here, should remain useful for benchmarking later model versions.</p><p>The clinical scope is also narrow. We tested perioperative consultations for a single pediatric malformation, so the model-specific conclusions need replication before they are applied to other settings; the evaluation structure (forced-ranking + citation audit + EAU-anchored safety audit) can be reused across other pediatric or adult specialties, as analogous multidimensional expert evaluations have proven viable in both adult urology and cross-disciplinary domains [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. However, the specific model rankings cannot.</p><p>Finally, the 8-reviewer safety audit assessed static written outputs, not real-time clinical interactions. A clinical deployment would need prospective patient outcome monitoring, formal red-flag detection, and institutional safety review [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref49">49</xref>], together with cross-cultural validation [<xref ref-type="bibr" rid="ref50">50</xref>] to test whether the dimension priorities (technical accuracy vs empathy) and the caregiver expert divergence pattern hold elsewhere. The evaluation framework, anchored by the V/PV/F/G/NR 5-category classification with dual-reviewer adjudication (Cohen &#x03BA;=0.702, substantial agreement) and canonical-source reverification of all 30 disagreements, is portable across clinical domains; the specific results are not.</p></sec><sec id="s4-5"><title>Implications</title><p>These implications rest on several methodological strengths: a dual-perspective design capturing complementary clinician- and caregiver-rated dimensions; double-blind forced-ranking with randomized labels to mitigate central-tendency bias [<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref46">46</xref>]; a 5-category consensus citation classification (V/PV/F/G/NR) with dual-reviewer adjudication (Cohen &#x03BA;=0.702) and canonical-source reverification; an 8-reviewer EAU 2025&#x2013;anchored clinical safety audit; robust cross-stratum ranking stability confirmed across sociodemographic subgroups, clinical risk tiers, and poststratification weighting (Spearman &#x03C1;=1.00); and a time-stamped, configuration-specific protocol (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) that bounds temporal validity and establishes a reproducible, unoptimized baseline.</p><p>This unoptimized baseline provides a systematic starting point for testing future prompt-engineering and retrieval-augmented interventions linked to the specific deficits we identified. Negative-constraint prompting&#x2014;instructing models to cite only verifiable literature or to declare &#x201C;insufficient evidence&#x201D;&#x2014;could test whether citation fabrication rates can be suppressed without degrading response quality [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. Few-shot prompting [<xref ref-type="bibr" rid="ref51">51</xref>] with standardized exemplar physician responses could target specific dimension-level deficits, such as ChatGPT-4o&#x2019;s low Empathy scores or OpenEvidence&#x2019;s poor Comprehensibility, provided these trials are conducted prospectively with concurrent evaluator recruitment to avoid recall bias. Retrieval-augmented generation (RAG) grounded in authoritative clinical resources (eg, EAU/AUA guidelines) is the most clinically actionable next step [<xref ref-type="bibr" rid="ref52">52</xref>]; the specific errors flagged in our safety audit&#x2014;off-indication testosterone recommendations, erroneous recovery timelines, and missing emergency-triage instructions&#x2014;are exactly the failure modes RAG architectures are designed to prevent. Future prospective studies evaluating these technical optimizations should also incorporate validated psychometric instruments, such as the eHEALS [<xref ref-type="bibr" rid="ref36">36</xref>] and the STAI [<xref ref-type="bibr" rid="ref47">47</xref>], at the time of evaluation to enable direct correlation of these psychological constructs with dimension-level model ratings.</p></sec><sec id="s4-6"><title>Conclusions</title><p>No single LLM was uniformly best for perioperative consultation in pediatric hypospadias. Beyond the specific rankings, which will date as models update, this study contributes 2 findings that should remain useful across model generations: a characterization of which evaluation dimensions clinicians and caregivers each prioritize, and direct evidence that bibliographic accuracy and clinical content safety are dissociable dimensions and need to be assessed separately. The accompanying evaluation framework&#x2014;dual-perspective forced-ranking, multidimensional rubric, independent citation verification, 8-reviewer guideline-anchored safety audit, and risk-stratified analysis&#x2014;provides a reproducible baseline against which future prompt-engineering and retrieval-augmented interventions can be tested. For perioperative uses, AI should be deployed under a tiered human-machine collaboration model, with mandatory clinician oversight in high-risk scenarios [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>].</p></sec></sec></body><back><ack><p>The authors express sincere gratitude to all caregivers and their children who participated in this study for their time, trust, and valuable feedback during the perioperative period. The authors are deeply grateful to the expert panel members for their rigorous evaluations and professional insights. Special thanks are extended to all members of the research team for their dedication to data collection, quality control, and coordination throughout the study. This work would not have been possible without the collective contributions of all participants and collaborators. The authors declare the use of generative artificial intelligence (GAI) in the research and writing process. According to the GAIDeT taxonomy (2025), the following tasks were delegated to GAI tools under full human supervision: proofreading and editing. The GAI tool used was DeepSeek (version 3.0; Beijing DeepSeek Artificial Intelligence Technology Co, Ltd). Responsibility for the final manuscript lies entirely with the authors. GAI tools are not listed as authors and do not bear responsibility for the final outcomes. All AI-assisted output was reviewed and edited by the authors, who take full responsibility for the integrity of the manuscript. The 5 large language models evaluated in this study (including DeepSeek) served as research subjects; the use of DeepSeek as a writing aid was conceptually and operationally separate from its role as one of the evaluated models and did not influence the blinded evaluation.</p></ack><notes><sec><title>Funding</title><p>This research was funded by the Postdoctoral Research Fund of West China Hospital, Sichuan University (grant 2025HXBH053). All authors declare that they have no financial interests or collaborative relationships with the developers of the artificial intelligence models involved in this study.</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed during this study are available from the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: CY, TK</p><p>Data curation: TK, CY</p><p>Formal analysis: TK, CY, XH, WH</p><p>Funding acquisition: CY</p><p>Investigation: TK (lead), XH (supporting), CY (supporting)</p><p>Methodology: TK, CY</p><p>Project administration: TK, CY, XH (supporting)</p><p>Resources: CY (lead), WH (supporting)</p><p>Supervision: CY, WH</p><p>Validation: TK, CY, WH (supporting)</p><p>Visualization: TK, CY</p><p>Writing &#x2013; original draft: TK, CY, XH, WH</p><p>Writing &#x2013; review &#x0026; editing: TK, CY, XH, WH</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AI</term><def><p>artificial intelligence</p></def></def-item><def-item><term id="abb2">EAU</term><def><p>European Association of Urology</p></def></def-item><def-item><term id="abb3">eHEALS</term><def><p>eHealth Literacy Scale</p></def></def-item><def-item><term id="abb4">F</term><def><p>Fabricated (citation category)</p></def></def-item><def-item><term id="abb5">G</term><def><p>Guideline-Based Non-Specific (citation category)</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">NR</term><def><p>No References (citation category)</p></def></def-item><def-item><term id="abb8">PV</term><def><p>Partially Verifiable (citation category)</p></def></def-item><def-item><term id="abb9">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb10">STAI</term><def><p>State-Trait Anxiety Inventory</p></def></def-item><def-item><term id="abb11">V</term><def><p>Verifiable (citation category)</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Springer</surname><given-names>A</given-names> </name><name name-style="western"><surname>van den Heijkant</surname><given-names>M</given-names> </name><name name-style="western"><surname>Baumann</surname><given-names>S</given-names> </name></person-group><article-title>Worldwide prevalence of hypospadias</article-title><source>J Pediatr Urol</source><year>2016</year><month>06</month><volume>12</volume><issue>3</issue><fpage>152</fpage><pub-id pub-id-type="doi">10.1016/j.jpurol.2015.12.002</pub-id><pub-id pub-id-type="medline">26810252</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gozar</surname><given-names>H</given-names> </name><name name-style="western"><surname>Bara</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Dicu</surname><given-names>E</given-names> </name><name name-style="western"><surname>Derzsi</surname><given-names>Z</given-names> </name></person-group><article-title>Current perspectives in hypospadias research: a scoping review of articles published in 2021 (Review)</article-title><source>Exp Ther Med</source><year>2023</year><month>05</month><volume>25</volume><issue>5</issue><fpage>211</fpage><pub-id pub-id-type="doi">10.3892/etm.2023.11910</pub-id><pub-id pub-id-type="medline">37090085</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rynja</surname><given-names>SP</given-names> </name><name name-style="western"><surname>de Jong</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bosch</surname><given-names>J</given-names> </name><name name-style="western"><surname>de Kort</surname><given-names>LMO</given-names> </name></person-group><article-title>Functional, cosmetic and psychosexual results in adult men who underwent hypospadias correction in childhood</article-title><source>J Pediatr Urol</source><year>2011</year><month>10</month><volume>7</volume><issue>5</issue><fpage>504</fpage><lpage>515</lpage><pub-id pub-id-type="doi">10.1016/j.jpurol.2011.02.008</pub-id><pub-id pub-id-type="medline">21429804</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bray</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sharpe</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gichuru</surname><given-names>P</given-names> </name><name name-style="western"><surname>Fortune</surname><given-names>PM</given-names> </name><name name-style="western"><surname>Blake</surname><given-names>L</given-names> </name><name name-style="western"><surname>Appleton</surname><given-names>V</given-names> </name></person-group><article-title>The acceptability and impact of the Xploro digital therapeutic platform to inform and prepare children for planned procedures in a hospital: before and after evaluation study</article-title><source>J Med Internet Res</source><year>2020</year><month>08</month><day>11</day><volume>22</volume><issue>8</issue><fpage>e17367</fpage><pub-id pub-id-type="doi">10.2196/17367</pub-id><pub-id pub-id-type="medline">32780025</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ming</surname><given-names>L</given-names> </name><name name-style="western"><surname>Geng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xianrong</surname><given-names>L</given-names> </name></person-group><article-title>Preoperative psychological status of children with hypospadias and their families [in Chinese]</article-title><source>Int J Nurs</source><year>2020</year><volume>39</volume><issue>2</issue><fpage>242</fpage><lpage>246</lpage><pub-id pub-id-type="doi">10.3760/cma.j.issn.1673-4351.2020.02.016</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kj&#x00E6;rulff</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Andersen</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Kingod</surname><given-names>N</given-names> </name><name name-style="western"><surname>Nex&#x00F8;</surname><given-names>MA</given-names> </name></person-group><article-title>When people with chronic conditions turn to peers on social media to obtain and share information: systematic review of the implications for relationships with health care professionals</article-title><source>J Med Internet Res</source><year>2023</year><month>04</month><day>17</day><volume>25</volume><fpage>e41156</fpage><pub-id pub-id-type="doi">10.2196/41156</pub-id><pub-id pub-id-type="medline">37067874</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bharathi Mohan</surname><given-names>G</given-names> </name><name name-style="western"><surname>Prasanna Kumar</surname><given-names>R</given-names> </name><name name-style="western"><surname>Vishal Krishh</surname><given-names>P</given-names> </name><etal/></person-group><article-title>An analysis of large language models: their impact and potential applications</article-title><source>Knowl Inf Syst</source><year>2024</year><month>09</month><volume>66</volume><issue>9</issue><fpage>5047</fpage><lpage>5070</lpage><pub-id pub-id-type="doi">10.1007/s10115-024-02120-8</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Nori</surname><given-names>H</given-names> </name><name name-style="western"><surname>King</surname><given-names>N</given-names> </name><name name-style="western"><surname>McKinney</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Carignan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Horvitz</surname><given-names>E</given-names> </name></person-group><article-title>Capabilities of GPT-4 on medical challenge problems</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 20, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.13375</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>D</given-names> </name><name name-style="western"><surname>Parsa</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hope</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Physician and artificial intelligence chatbot responses to cancer questions from social media</article-title><source>JAMA Oncol</source><year>2024</year><month>07</month><day>1</day><volume>10</volume><issue>7</issue><fpage>956</fpage><lpage>960</lpage><pub-id pub-id-type="doi">10.1001/jamaoncol.2024.0836</pub-id><pub-id pub-id-type="medline">38753317</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bernstein</surname><given-names>IA</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>YV</given-names> </name><name name-style="western"><surname>Govil</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Comparison of ophthalmologist and large language model chatbot responses to online patient eye care questions</article-title><source>JAMA Netw Open</source><year>2023</year><month>08</month><day>1</day><volume>6</volume><issue>8</issue><fpage>e2330320</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2023.30320</pub-id><pub-id pub-id-type="medline">37606922</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yau</surname><given-names>JYS</given-names> </name><name name-style="western"><surname>Saadat</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hsu</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Accuracy of prospective assessments of 4 large language model chatbot responses to patient questions about emergency care: experimental comparative study</article-title><source>J Med Internet Res</source><year>2024</year><month>11</month><day>4</day><volume>26</volume><fpage>e60291</fpage><pub-id pub-id-type="doi">10.2196/60291</pub-id><pub-id pub-id-type="medline">39496149</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Athaluri</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Manthena</surname><given-names>SV</given-names> </name><name name-style="western"><surname>Kesapragada</surname><given-names>V</given-names> </name><name name-style="western"><surname>Yarlagadda</surname><given-names>V</given-names> </name><name name-style="western"><surname>Dave</surname><given-names>T</given-names> </name><name name-style="western"><surname>Duddumpudi</surname><given-names>RTS</given-names> </name></person-group><article-title>Exploring the boundaries of reality: investigating the phenomenon of artificial intelligence hallucination in scientific writing through ChatGPT references</article-title><source>Cureus</source><year>2023</year><month>04</month><volume>15</volume><issue>4</issue><fpage>e37432</fpage><pub-id pub-id-type="doi">10.7759/cureus.37432</pub-id><pub-id pub-id-type="medline">37182055</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alkaissi</surname><given-names>H</given-names> </name><name name-style="western"><surname>McFarlane</surname><given-names>SI</given-names> </name></person-group><article-title>Artificial hallucinations in ChatGPT: implications in scientific writing</article-title><source>Cureus</source><year>2023</year><month>02</month><volume>15</volume><issue>2</issue><fpage>e35179</fpage><pub-id pub-id-type="doi">10.7759/cureus.35179</pub-id><pub-id pub-id-type="medline">36811129</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cung</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sosa</surname><given-names>B</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>HS</given-names> </name><etal/></person-group><article-title>The performance of artificial intelligence chatbot large language models to address skeletal biology and bone health queries</article-title><source>J Bone Miner Res</source><year>2024</year><month>03</month><day>22</day><volume>39</volume><issue>2</issue><fpage>106</fpage><lpage>115</lpage><pub-id pub-id-type="doi">10.1093/jbmr/zjad007</pub-id><pub-id pub-id-type="medline">38477743</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Faraj</surname><given-names>S</given-names> </name><name name-style="western"><surname>Clermidi</surname><given-names>P</given-names> </name><name name-style="western"><surname>Irtan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Madec</surname><given-names>FX</given-names> </name></person-group><article-title>Artificial intelligence chatbots vs. YPUC pediatric urologists: performance on a Campbell Walsh urology hypospadiology questionnaire</article-title><source>World J Urol</source><year>2025</year><month>11</month><day>24</day><volume>43</volume><issue>1</issue><fpage>719</fpage><pub-id pub-id-type="doi">10.1007/s00345-025-06104-3</pub-id><pub-id pub-id-type="medline">41284104</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Temsah</surname><given-names>A</given-names> </name><name name-style="western"><surname>Alhasan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Altamimi</surname><given-names>I</given-names> </name><etal/></person-group><article-title>DeepSeek in healthcare: revealing opportunities and steering challenges of a new open-source artificial Intelligence frontier</article-title><source>Cureus</source><year>2025</year><month>02</month><volume>17</volume><issue>2</issue><fpage>e79221</fpage><pub-id pub-id-type="doi">10.7759/cureus.79221</pub-id><pub-id pub-id-type="medline">39974299</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sezgin</surname><given-names>E</given-names> </name><name name-style="western"><surname>Jackson</surname><given-names>DI</given-names> </name><name name-style="western"><surname>Kocaballi</surname><given-names>AB</given-names> </name><etal/></person-group><article-title>Can large language models aid caregivers of pediatric cancer patients in information seeking? A cross-sectional investigation</article-title><source>Cancer Med</source><year>2025</year><month>01</month><volume>14</volume><issue>1</issue><fpage>e70554</fpage><pub-id pub-id-type="doi">10.1002/cam4.70554</pub-id><pub-id pub-id-type="medline">39776222</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="web"><article-title>Statistical communiqu&#x00E9; of the People&#x2019;s Republic of China on the 2024 national economic and social development [in Chinese]</article-title><source>National Bureau of Statistics of China</source><year>2025</year><access-date>2025-05-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.stats.gov.cn/xxgk/sjfb/zxfb2020/202502/t20250228_1958817.html">https://www.stats.gov.cn/xxgk/sjfb/zxfb2020/202502/t20250228_1958817.html</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huo</surname><given-names>B</given-names> </name><name name-style="western"><surname>Collins</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chartash</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Reporting guideline for Chatbot Health Advice studies: the CHART statement</article-title><source>BMC Med</source><year>2025</year><month>08</month><day>1</day><volume>23</volume><issue>1</issue><fpage>447</fpage><pub-id pub-id-type="doi">10.1186/s12916-025-04274-w</pub-id><pub-id pub-id-type="medline">40745595</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="web"><article-title>2024-2025 China AI large model market status and development trend research report</article-title><source>iiMedia Research Group</source><year>2024</year><access-date>2025-03-20</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.iimedia.cn/c400/104152.html">https://www.iimedia.cn/c400/104152.html</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Maslej</surname><given-names>N</given-names> </name><name name-style="western"><surname>Fattorini</surname><given-names>L</given-names> </name><name name-style="western"><surname>Perrault</surname><given-names>R</given-names> </name><etal/></person-group><article-title>The AI Index 2024 Annual Report</article-title><source>Stanford HAI</source><year>2024</year><access-date>2025-03-20</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://aiindex.stanford.edu/report">https://aiindex.stanford.edu/report</ext-link></comment></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="web"><article-title>EAU guidelines on paediatric urology</article-title><source>European Association of Urology</source><year>2025</year><access-date>2026-05-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://uroweb.org/guidelines/paediatric-urology">https://uroweb.org/guidelines/paediatric-urology</ext-link></comment></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Davidson</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Disma</surname><given-names>N</given-names> </name><name name-style="western"><surname>de Graaff</surname><given-names>JC</given-names> </name><etal/></person-group><article-title>Neurodevelopmental outcome at 2 years of age after general anaesthesia and awake-regional anaesthesia in infancy (GAS): an international multicentre, randomised controlled trial</article-title><source>Lancet</source><year>2016</year><month>01</month><day>16</day><volume>387</volume><issue>10015</issue><fpage>239</fpage><lpage>250</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(15)00608-X</pub-id><pub-id pub-id-type="medline">26507180</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>LS</given-names> </name><name name-style="western"><surname>Li</surname><given-names>G</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>TLK</given-names> </name><etal/></person-group><article-title>Association between a single general anesthesia exposure before age 36 months and neurocognitive outcomes in later childhood</article-title><source>JAMA</source><year>2016</year><month>06</month><day>7</day><volume>315</volume><issue>21</issue><fpage>2312</fpage><pub-id pub-id-type="doi">10.1001/jama.2016.6967</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Warner</surname><given-names>DO</given-names> </name><name name-style="western"><surname>Zaccariello</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Katusic</surname><given-names>SK</given-names> </name><etal/></person-group><article-title>Neuropsychological and behavioral outcomes after exposure of young children to procedures requiring general anesthesia</article-title><source>Anesthesiology</source><year>2018</year><volume>129</volume><issue>1</issue><fpage>89</fpage><lpage>105</lpage><pub-id pub-id-type="doi">10.1097/ALN.0000000000002232</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gajjar</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>RP</given-names> </name><name name-style="western"><surname>Paliwoda</surname><given-names>ED</given-names> </name><etal/></person-group><article-title>Usefulness and accuracy of artificial intelligence chatbot responses to patient questions for neurosurgical procedures</article-title><source>Neurosurgery</source><year>2024</year><month>02</month><day>14</day><pub-id pub-id-type="doi">10.1227/neu.0000000000002856</pub-id><pub-id pub-id-type="medline">38353558</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Evaluation of large language models in breast cancer clinical scenarios: a comparative analysis based on ChatGPT-3.5, ChatGPT-4.0, and Claude2</article-title><source>Int J Surg</source><year>2024</year><month>04</month><day>1</day><volume>110</volume><issue>4</issue><fpage>1941</fpage><lpage>1950</lpage><pub-id pub-id-type="doi">10.1097/JS9.0000000000001066</pub-id><pub-id pub-id-type="medline">38668655</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vishnevetsky</surname><given-names>J</given-names> </name><name name-style="western"><surname>Walters</surname><given-names>CB</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>KS</given-names> </name></person-group><article-title>Interrater reliability of the Patient Education Materials Assessment Tool (PEMAT)</article-title><source>Patient Educ Couns</source><year>2018</year><month>03</month><volume>101</volume><issue>3</issue><fpage>490</fpage><lpage>496</lpage><pub-id pub-id-type="doi">10.1016/j.pec.2017.09.003</pub-id><pub-id pub-id-type="medline">28899713</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shoemaker</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Wolf</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Brach</surname><given-names>C</given-names> </name></person-group><article-title>Development of the Patient Education Materials Assessment Tool (PEMAT): a new measure of understandability and actionability for print and audiovisual patient information</article-title><source>Patient Educ Couns</source><year>2014</year><month>09</month><volume>96</volume><issue>3</issue><fpage>395</fpage><lpage>403</lpage><pub-id pub-id-type="doi">10.1016/j.pec.2014.05.027</pub-id><pub-id pub-id-type="medline">24973195</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>ZC</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>HS</given-names> </name><etal/></person-group><article-title>Analysis of the social and clinical factors affecting the age of children when receiving surgery for hypospadias: a retrospective study of 1611 cases in a single center</article-title><source>Asian J Androl</source><year>2021</year><volume>23</volume><issue>5</issue><fpage>527</fpage><lpage>531</lpage><pub-id pub-id-type="doi">10.4103/aja.aja_11_21</pub-id><pub-id pub-id-type="medline">33723097</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>H</given-names> </name><etal/></person-group><article-title>The role of sociodemographic factors in family resilience of Chinese children with chronic illnesses: a cross-sectional study</article-title><source>Child Care Health Dev</source><year>2025</year><month>11</month><volume>51</volume><issue>6</issue><fpage>e70173</fpage><pub-id pub-id-type="doi">10.1111/cch.70173</pub-id><pub-id pub-id-type="medline">41161706</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name></person-group><source>Statistical Power Analysis for the Behavioral Sciences</source><year>1988</year><edition>2</edition><publisher-name>Lawrence Erlbaum Associates</publisher-name><pub-id pub-id-type="other">0-8058-0283-5</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chesser</surname><given-names>A</given-names> </name><name name-style="western"><surname>Burke</surname><given-names>A</given-names> </name><name name-style="western"><surname>Reyes</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rohrberg</surname><given-names>T</given-names> </name></person-group><article-title>Navigating the digital divide: a systematic review of eHealth literacy in underserved populations in the United States</article-title><source>Inform Health Soc Care</source><year>2016</year><volume>41</volume><issue>1</issue><fpage>1</fpage><lpage>19</lpage><pub-id pub-id-type="doi">10.3109/17538157.2014.948171</pub-id><pub-id pub-id-type="medline">25710808</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mackert</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mabry-Flynn</surname><given-names>A</given-names> </name><name name-style="western"><surname>Champlin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Donovan</surname><given-names>EE</given-names> </name><name name-style="western"><surname>Pounders</surname><given-names>K</given-names> </name></person-group><article-title>Health literacy and health information technology adoption: the potential for a new digital divide</article-title><source>J Med Internet Res</source><year>2016</year><month>10</month><day>4</day><volume>18</volume><issue>10</issue><fpage>e264</fpage><pub-id pub-id-type="doi">10.2196/jmir.6349</pub-id><pub-id pub-id-type="medline">27702738</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naghdi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>P</given-names> </name><name name-style="western"><surname>Essers</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Artificial intelligence-simplified information to advance reproductive genetic literacy and health equity</article-title><source>Hum Reprod</source><year>2025</year><month>09</month><day>1</day><volume>40</volume><issue>9</issue><fpage>1681</fpage><lpage>1688</lpage><pub-id pub-id-type="doi">10.1093/humrep/deaf135</pub-id><pub-id pub-id-type="medline">40692125</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Norman</surname><given-names>CD</given-names> </name><name name-style="western"><surname>Skinner</surname><given-names>HA</given-names> </name></person-group><article-title>eHealth literacy: essential skills for consumer health in a networked world</article-title><source>J Med Internet Res</source><year>2006</year><month>06</month><day>16</day><volume>8</volume><issue>2</issue><fpage>e9</fpage><pub-id pub-id-type="doi">10.2196/jmir.8.2.e9</pub-id><pub-id pub-id-type="medline">16867972</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Masoud</surname><given-names>RI</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ferianc</surname><given-names>M</given-names> </name><name name-style="western"><surname>Treleaven</surname><given-names>P</given-names> </name><name name-style="western"><surname>Rodrigues</surname><given-names>M</given-names> </name></person-group><article-title>Cultural alignment in large language models: an explanatory analysis based on Hofstede&#x2019;s cultural dimensions</article-title><access-date>2026-07-14</access-date><conf-name>Proceedings of the 31st International Conference on Computational Linguistics (COLING)</conf-name><conf-date>Jan 19-24, 2025</conf-date><conf-loc>Abu Dhabi, UAE</conf-loc><fpage>8474</fpage><lpage>8503</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.coling-main.567/">https://aclanthology.org/2025.coling-main.567/</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mesk&#x00F3;</surname><given-names>B</given-names> </name><name name-style="western"><surname>Topol</surname><given-names>EJ</given-names> </name></person-group><article-title>The imperative for regulatory oversight of large language models (or generative AI) in healthcare</article-title><source>NPJ Digit Med</source><year>2023</year><month>07</month><day>6</day><volume>6</volume><issue>1</issue><fpage>120</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00873-0</pub-id><pub-id pub-id-type="medline">37414860</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Topol</surname><given-names>EJ</given-names> </name></person-group><article-title>High-performance medicine: the convergence of human and artificial intelligence</article-title><source>Nat Med</source><year>2019</year><month>01</month><volume>25</volume><issue>1</issue><fpage>44</fpage><lpage>56</lpage><pub-id pub-id-type="doi">10.1038/s41591-018-0300-7</pub-id><pub-id pub-id-type="medline">30617339</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bubeck</surname><given-names>S</given-names> </name><name name-style="western"><surname>Petro</surname><given-names>JB</given-names> </name></person-group><article-title>Benefits, limits, and risks of GPT-4 as an AI chatbot for medicine</article-title><source>N Engl J Med</source><year>2023</year><month>03</month><day>30</day><volume>388</volume><issue>13</issue><fpage>1233</fpage><lpage>1239</lpage><pub-id pub-id-type="doi">10.1056/NEJMsr2214184</pub-id><pub-id pub-id-type="medline">36988602</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liyanage</surname><given-names>H</given-names> </name><name name-style="western"><surname>Liaw</surname><given-names>ST</given-names> </name><name name-style="western"><surname>Jonnagaddala</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Artificial intelligence in primary health care: perceptions, issues, and challenges</article-title><source>Yearb Med Inform</source><year>2019</year><month>08</month><volume>28</volume><issue>1</issue><fpage>41</fpage><lpage>46</lpage><pub-id pub-id-type="doi">10.1055/s-0039-1677901</pub-id><pub-id pub-id-type="medline">31022751</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Krajcer</surname><given-names>Z</given-names> </name></person-group><article-title>Artificial intelligence for education, proctoring, and credentialing in cardiovascular medicine</article-title><source>Tex Heart Inst J</source><year>2022</year><month>03</month><day>1</day><volume>49</volume><issue>2</issue><fpage>e217572</fpage><pub-id pub-id-type="doi">10.14503/THIJ-21-7572</pub-id><pub-id pub-id-type="medline">35481865</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>H</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Evaluating the performance of different large language models on health consultation and patient education in urolithiasis</article-title><source>J Med Syst</source><year>2023</year><month>11</month><day>24</day><volume>47</volume><issue>1</issue><fpage>125</fpage><pub-id pub-id-type="doi">10.1007/s10916-023-02021-3</pub-id><pub-id pub-id-type="medline">37999899</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nichani</surname><given-names>PAH</given-names> </name><name name-style="western"><surname>Ong Tone</surname><given-names>S</given-names> </name><name name-style="western"><surname>AlShaker</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Teichman</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>CC</given-names> </name></person-group><article-title>Use of online large language model chatbots in cornea clinics</article-title><source>Cornea</source><year>2024</year><month>12</month><day>3</day><volume>44</volume><issue>6</issue><fpage>788</fpage><lpage>794</lpage><pub-id pub-id-type="doi">10.1097/ICO.0000000000003747</pub-id><pub-id pub-id-type="medline">39625129</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>B&#x00F6;ckenholt</surname><given-names>U</given-names> </name></person-group><article-title>Comparative judgments as an alternative to ratings: identifying the scale origin</article-title><source>Psychol Methods</source><year>2004</year><month>12</month><volume>9</volume><issue>4</issue><fpage>453</fpage><lpage>465</lpage><pub-id pub-id-type="doi">10.1037/1082-989X.9.4.453</pub-id><pub-id pub-id-type="medline">15598099</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Krosnick</surname><given-names>JA</given-names> </name></person-group><article-title>Response strategies for coping with the cognitive demands of attitude measures in surveys</article-title><source>Appl Cogn Psychol</source><year>1991</year><month>05</month><volume>5</volume><issue>3</issue><fpage>213</fpage><lpage>236</lpage><pub-id pub-id-type="doi">10.1002/acp.2350050305</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Spielberger</surname><given-names>CD</given-names> </name><name name-style="western"><surname>Gorsuch</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Lushene</surname><given-names>R</given-names> </name><name name-style="western"><surname>Vagg</surname><given-names>PR</given-names> </name><name name-style="western"><surname>Jacobs</surname><given-names>GA</given-names> </name></person-group><source>Manual for the State-Trait Anxiety Inventory</source><year>1983</year><publisher-name>Consulting Psychologists Press</publisher-name></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zaharia</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>J</given-names> </name></person-group><article-title>How Is ChatGPT&#x2019;s behavior changing over time?</article-title><source>Harvard Data Sci Rev</source><year>2024</year><volume>6</volume><issue>2</issue><pub-id pub-id-type="doi">10.1162/99608f92.5317da47</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vasey</surname><given-names>B</given-names> </name><name name-style="western"><surname>Nagendran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Campbell</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Reporting guideline for the early-stage clinical evaluation of decision support systems driven by artificial intelligence: DECIDE-AI</article-title><source>Nat Med</source><year>2022</year><month>05</month><volume>28</volume><issue>5</issue><fpage>924</fpage><lpage>933</lpage><pub-id pub-id-type="doi">10.1038/s41591-022-01772-9</pub-id><pub-id pub-id-type="medline">35585198</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="web"><article-title>Ethics and governance of artificial intelligence for health</article-title><source>World Health Organization</source><year>2021</year><access-date>2026-06-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/publications/i/item/9789240029200">https://www.who.int/publications/i/item/9789240029200</ext-link></comment></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Mann</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ryder</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Language models are few-shot learners</article-title><year>2020</year><access-date>2026-07-15</access-date><conf-name>Proceedings of the 34th International Conference on Neural Information Processing Systems (NeurIPS)</conf-name><conf-date>Dec 6-12, 2020</conf-date><conf-loc>Vancouver BC Canada</conf-loc><fpage>1877</fpage><lpage>1901</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2005.14165">https://arxiv.org/abs/2005.14165</ext-link></comment></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>P</given-names> </name><name name-style="western"><surname>Perez</surname><given-names>E</given-names> </name><name name-style="western"><surname>Piktus</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation for knowledge-intensive NLP tasks</article-title><access-date>2026-07-15</access-date><conf-name>Proceedings of the 34th International Conference on Neural Information Processing Systems (NeurIPS)</conf-name><conf-date>Dec 6-12, 2020</conf-date><conf-loc>Vancouver BC Canada</conf-loc><fpage>9459</fpage><lpage>9474</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2005.11401">https://arxiv.org/abs/2005.11401</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Large language model access protocol.</p><media xlink:href="jmir_v28i1e93393_app1.docx" xlink:title="DOCX File, 17 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Reference authenticity verification&#x2014;summary findings.</p><media xlink:href="jmir_v28i1e93393_app2.docx" xlink:title="DOCX File, 16 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Eight-reviewer clinical safety audit&#x2014;per-reviewer findings.</p><media xlink:href="jmir_v28i1e93393_app3.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Complete 40-question bank.</p><media xlink:href="jmir_v28i1e93393_app4.pdf" xlink:title="PDF File, 326 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>PlainText_LLM_Responses.</p><media xlink:href="jmir_v28i1e93393_app5.pdf" xlink:title="PDF File, 1100 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6</label><p>Evaluation dimension definitions and scoring procedure.</p><media xlink:href="jmir_v28i1e93393_app6.docx" xlink:title="DOCX File, 14 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7</label><p>Complete pairwise comparison tables.</p><media xlink:href="jmir_v28i1e93393_app7.pdf" xlink:title="PDF File, 137 KB"/></supplementary-material><supplementary-material id="app8"><label>Multimedia Appendix 8</label><p>Theme-based analysis.</p><media xlink:href="jmir_v28i1e93393_app8.pdf" xlink:title="PDF File, 271 KB"/></supplementary-material><supplementary-material id="app9"><label>Multimedia Appendix 9</label><p>Risk-level stratified analysis.</p><media xlink:href="jmir_v28i1e93393_app9.pdf" xlink:title="PDF File, 1016 KB"/></supplementary-material><supplementary-material id="app10"><label>Multimedia Appendix 10</label><p>Demographic subgroup stratified analysis.</p><media xlink:href="jmir_v28i1e93393_app10.pdf" xlink:title="PDF File, 859 KB"/></supplementary-material><supplementary-material id="app11"><label>Multimedia Appendix 11</label><p>Per-question expert ratings.</p><media xlink:href="jmir_v28i1e93393_app11.pdf" xlink:title="PDF File, 15924 KB"/></supplementary-material><supplementary-material id="app12"><label>Multimedia Appendix 12</label><p>Per-question caregiver ratings.</p><media xlink:href="jmir_v28i1e93393_app12.pdf" xlink:title="PDF File, 8194 KB"/></supplementary-material><supplementary-material id="app13"><label>Multimedia Appendix 13</label><p>European Association of Urology 2025 cross-check.</p><media xlink:href="jmir_v28i1e93393_app13.pdf" xlink:title="PDF File, 169 KB"/></supplementary-material><supplementary-material id="app14"><label>Checklist 1</label><p>CHART checklist.</p><media xlink:href="jmir_v28i1e93393_app14.docx" xlink:title="DOCX File, 16 KB"/></supplementary-material></app-group></back></article>