<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e95162</article-id><article-id pub-id-type="doi">10.2196/95162</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Evaluating a Guideline-Integrated Clinical Interaction Framework Vs a Standard Large Language Model Interaction for Dietary Recommendations in Recurrent Urolithiasis: In Silico Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Wang</surname><given-names>Xiaofeng</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Li</surname><given-names>Jun</given-names></name><degrees>BM</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hu</surname><given-names>Yudong</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Yujie</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhong</surname><given-names>Yong</given-names></name><degrees>BM</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhu</surname><given-names>Faming</given-names></name><degrees>BM</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yuan</surname><given-names>Ye</given-names></name><degrees>BM</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yang</surname><given-names>Fan</given-names></name><degrees>BM</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Ye</surname><given-names>Jin</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>Department of Urology, The Thirteenth People&#x2019;s Hospital</institution><addr-line>No. 16, Railway New Village, Huangjueping Subdistrict, Jiulongpo District</addr-line><addr-line>Chongqing</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Ferguson</surname><given-names>Frederick</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Marshall</surname><given-names>Robert</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Jin Ye, MD, Department of Urology, The Thirteenth People&#x2019;s Hospital, No. 16, Railway New Village, Huangjueping Subdistrict, Jiulongpo District, Chongqing, China, 86 15808005558; <email>yoko607@163.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>4</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e95162</elocation-id><history><date date-type="received"><day>11</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>14</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>18</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Xiaofeng Wang, Jun Li, Yudong Hu, Yujie Chen, Yong Zhong, Faming Zhu, Ye Yuan, Fan Yang, Jin Ye. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 4.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e95162"/><abstract><sec><title>Background</title><p>Personalized dietary counseling is central to recurrence prevention in patients with urolithiasis, particularly after a 24-hour urine metabolic evaluation. However, translating quantitative metabolic abnormalities into patient-facing, guideline-concordant, and safe dietary recommendations can be challenging in routine clinical practice. Large language models (LLMs) may assist with this task, but unguided responses may overlook key metabolic priorities or case-specific safety constraints.</p></sec><sec><title>Objective</title><p>This study evaluated whether a guideline-integrated, safety-aware, LLM-based clinical interaction framework (StoneAgent) could generate higher-quality, personalized dietary recommendations than a standard LLM configuration for recurrent urolithiasis. We also assessed whether any performance advantage persisted when the same clinical scenarios were presented as patient query&#x2013;style inputs.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted an in silico comparative study using 30 synthetic clinical vignettes representing common, mixed, and safety-relevant metabolic stone scenarios. For the primary experiment, StoneAgent and a standard LLM configuration were compared using structured vignette inputs. For the robustness experiment, each vignette was reformulated into 2 patient query&#x2013;style variants (query A and query B), preserving the same clinical content in more natural conversational language. A vignette-specific expert reference standard was developed from guideline-informed specialist consensus. Three independent reviewers blindly rated outputs on a 5-point Likert scale for metabolic specificity, guideline adherence, and actionability; safety was assessed as a binary outcome. For the patient query&#x2013;style experiment, query A and query B were aggregated at the vignette level for paired comparison.</p></sec><sec sec-type="results"><title>Results</title><p>In the structured-input experiment, StoneAgent achieved higher performance than the standard LLM across metabolic specificity, guideline adherence, and actionability, with median case-level scores of 5.00 (IQR 5.00&#x2010;5.00) vs 3.00 (IQR 2.75&#x2010;3.92) for metabolic specificity, 5.00 (IQR 5.00&#x2010;5.00) vs 3.67 (IQR 3.08&#x2010;4.00) for guideline adherence, and 5.00 (IQR 5.00&#x2010;5.00) vs 3.00 (IQR 2.67&#x2010;3.33) for actionability (all <italic>P</italic>&#x003C;.001). Safety pass rates were 100% (30/30) for StoneAgent and 83.3% (25/30) for the standard LLM (exact McNemar <italic>P</italic>=.06). In the patient query&#x2013;style robustness experiment, StoneAgent retained a directional advantage after case-level aggregation, with mean scores of 4.44 vs 3.62 for metabolic specificity, 4.51 vs 3.63 for guideline adherence, and 4.11 vs 3.40 for actionability. Safety pass rates were 100% (30/30) for StoneAgent and 90% (27/30) for the standard LLM (exact McNemar <italic>P</italic>=.25). The performance gap was more conservative under patient query&#x2013;style inputs compared with structured inputs, but the overall pattern remained consistent across domains.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In this in silico study, a guideline-integrated, safety-aware clinical interaction framework generated higher-quality dietary recommendations for recurrent urolithiasis than a standard LLM condition with structured vignette inputs. This advantage was retained with patient query&#x2013;style inputs. These findings suggest that explicit clinical framing, guideline grounding, and safety-oriented response scaffolding may improve the reliability of specialty counseling tasks involving metabolic stone prevention. Further validation is needed using real patient-authored queries and prospective clinical workflows.</p></sec></abstract><kwd-group><kwd>urolithiasis</kwd><kwd>kidney stones</kwd><kwd>large language models</kwd><kwd>AI</kwd><kwd>dietary counseling</kwd><kwd>metabolic evaluation</kwd><kwd>24-hour urine analysis</kwd><kwd>synthetic clinical vignettes</kwd><kwd>in silico evaluation</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Urolithiasis is a common condition worldwide, affecting roughly 7% to 13% of adults in North America and Europe [<xref ref-type="bibr" rid="ref1">1</xref>]. Nearly half of patients will develop another stone within 10 years of the initial diagnosis [<xref ref-type="bibr" rid="ref2">2</xref>]. Modern surgical techniques, including flexible ureteroscopy, have made the treatment of acute stone episodes far more effective. However, surgery primarily removes existing stones and does not correct the metabolic factors responsible for stone formation [<xref ref-type="bibr" rid="ref3">3</xref>]. For many patients, this means that the cycle of stone formation and intervention continues over time. Recurrent procedures are therefore common and are associated with both reduced quality of life and increased health care costs [<xref ref-type="bibr" rid="ref4">4</xref>]. Because of this, long-term management now places growing emphasis on prevention rather than repeated surgical treatment alone [<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>Clinical practice guidelines from the European Association of Urology (EAU) [<xref ref-type="bibr" rid="ref6">6</xref>] and the Canadian Urological Association (CUA) [<xref ref-type="bibr" rid="ref7">7</xref>] recommend metabolic evaluation for patients with recurrent or high-risk stones. In most cases, this involves 24-hour urine testing to detect metabolic abnormalities linked to stone formation, such as hypercalciuria, hypocitraturia, hyperoxaluria, and abnormal urinary pH [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. These metabolic findings can guide individualized dietary and lifestyle recommendations aimed at reducing recurrence [<xref ref-type="bibr" rid="ref10">10</xref>]. In reality, however, applying these recommendations consistently in routine practice is not straightforward. Interpreting metabolic profiles requires expertise, and detailed dietary counseling can be difficult to provide in busy clinical settings [<xref ref-type="bibr" rid="ref7">7</xref>]. As a result, many patients receive only general advice, for example, increasing fluid intake without recommendations that specifically target their metabolic risk factors [<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>Large language models (LLMs), including systems such as OpenAI&#x2019;s ChatGPT, have shown promise in medical communication and decision support, including patient education and the summarization of complex clinical information [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. However, the task addressed in recurrent urolithiasis is not simply to generate general dietary advice; it requires the model to interpret quantitative metabolic findings, prioritize the dominant stone-risk pattern, and translate those findings into recommendations that remain consistent with guideline principles and clinically safe in the presence of comorbid conditions. In this setting, unconstrained LLM responses may appear plausible while omitting key metabolic targets, misprioritizing the main preventive strategy, or overlooking case-specific safety constraints [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>An additional challenge is that real-world counseling requests are rarely presented as neatly structured case summaries. Patients typically ask questions in conversational language, with variable ordering of information and uneven emphasis on laboratory findings, symptoms, and concerns. As a result, performance observed under standardized vignette inputs may overestimate how reliably an LLM-based counseling framework can generalize to more naturalistic consultation scenarios. Evaluating robustness to patient query&#x2013;style inputs is therefore important when assessing the practical usefulness of guideline-integrated LLM-based counseling approaches.</p><p>In this study, we evaluated a guideline-integrated, safety-aware, LLM-based clinical interaction framework (StoneAgent) for generating personalized dietary recommendations in recurrent urolithiasis. We first compared StoneAgent with a standard LLM configuration using structured synthetic clinical vignettes designed to reflect common and safety-relevant metabolic stone scenarios. We then performed a robustness experiment using patient query&#x2013;style variants derived from the same vignettes to assess whether any performance advantage was retained under more naturalistic conversational inputs. We hypothesized that, compared with a standard unguided LLM interaction, StoneAgent would produce recommendations that were more metabolically specific, more concordant with guideline-based care, more actionable for patients, and safer in clinically constrained scenarios.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Dataset</title><sec id="s2-1-1"><title>Study Design</title><p>This in silico comparative study consisted of 2 linked experimental phases. In the primary phase, StoneAgent and a standard LLM configuration (GPT-5.1) were compared using structured synthetic clinical vignettes containing standardized clinical context and 24-hour urine metabolic data. In the second phase, each vignette was reformulated into 2 patient query&#x2013;style variants to evaluate whether the relative performance of the 2 frameworks was preserved under more natural conversational inputs. The same vignette-specific expert reference standard was used as the scoring anchor for both phases.</p><p>Thirty synthetic clinical vignettes were purposively constructed based on EAU and CUA guideline-relevant metabolic stone scenarios [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>] and stratified into categories A to C. In both phases, outputs from StoneAgent and the standard LLM were evaluated by 3 blinded reviewers for metabolic specificity, guideline adherence, and actionability using a 5-point Likert scale, while safety was evaluated separately as a binary outcome. In the query-style experiment, the 2 patient-style variants derived from the same vignette were aggregated at the vignette level for the primary paired comparison. The overall workflow is summarized in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study workflow diagram. CUA: Canadian Urological Association; EAU: European Association of Urology; LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95162_fig01.png"/></fig></sec><sec id="s2-1-2"><title>Ethical Considerations</title><p>This study used synthetic clinical vignettes generated specifically for research purposes and did not involve real patients, identifiable health information, clinical records, or interactions with human participants. Institutional review board approval was not required because the study did not involve human participants or access to clinical data. Informed consent was not applicable because no human participant data were collected or analyzed.</p></sec><sec id="s2-1-3"><title>Dataset</title><p>We constructed a dataset of 30 synthetic clinical vignettes to cover a broad range of metabolic patterns described in the EAU and CUA guidelines for urolithiasis [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Before analysis, the vignettes were reviewed for content validity by a senior urologist (&#x003E;15 y of experience) who did not participate in scoring. The reviewer checked that (1) each vignette reflected plausible clinical presentations and (2) the metabolic profiles were physiologically consistent. Ambiguous or unrealistic elements were revised accordingly. The full vignette dataset is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>Each vignette included 3 layers of information:</p><list list-type="order"><list-item><p>Patient characteristics: age, sex, BMI, and comorbidities (eg, hypertension, type 2 diabetes, and chronic kidney disease)</p></list-item><list-item><p>Stone history: recurrence pattern, reported stone composition (eg, calcium oxalate, uric acid, and struvite), and prior procedures</p></list-item><list-item><p>A 24-hour urine profile includes urine volume and quantitative parameters, including calcium, sodium, oxalate, citrate, uric acid, magnesium, and urine pH</p></list-item></list></sec><sec id="s2-1-4"><title>Cohort Stratification</title><p>To reflect differences in clinical complexity, the 30 vignettes were stratified into 3 prespecified categories: (1) single metabolic abnormality, (2) mixed or complex metabolic abnormalities, and (3) safety-critical scenarios in which otherwise standard dietary recommendations could become inappropriate because of patient-specific clinical factors. The comorbid conditions included in the safety-critical category were selected based on their potential to alter routine dietary counseling in recurrent stone prevention, including chronic kidney disease, heart failure, pregnancy, infection-related stones, primary hyperparathyroidism, and age-related functional limitations affecting hydration advice. These conditions were chosen because they represent clinically relevant situations in which generic dietary recommendations may require modification, additional caution, or prioritization of alternative management strategies. We focused on representative high-impact scenarios rather than attempting to reproduce the full spectrum of multimorbidity encountered in clinical practice. The distribution and key features of these vignettes are summarized in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Characteristics and stratification of the 30 synthetic clinical vignettes used in the study (N=30)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Category and clinical phenotype</td><td align="left" valign="bottom">Defining features</td><td align="left" valign="bottom">Number of cases (n)</td><td align="left" valign="bottom">Example cases</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Single metabolic abnormality (n=12)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sodium-dependent hypercalciuria</td><td align="left" valign="top">Urine Ca (&#x003E;8.0 mmol/d) driven by high Na (&#x003E;200 mmol/d); normal oxalate</td><td align="char" char="." valign="top">4</td><td align="left" valign="top">Case ID: C01, C12, C26</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Dietary hyperoxaluria</td><td align="left" valign="top">Urine oxalate &#x003E;0.5 mmol/d; high intake of oxalate-rich foods (eg, spinach, nuts)</td><td align="left" valign="top">3</td><td align="left" valign="top">Case ID: C04, C29</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hypocitraturia</td><td align="left" valign="top">Urine citrate &#x003C;1.5 mmol/d; often associated with low fruit/veg intake or high acid load</td><td align="left" valign="top">3</td><td align="left" valign="top">Case ID: C03, C13, C17</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Uric acid stones</td><td align="left" valign="top">Persistently low urine pH (&#x003C;5.5); hyperuricosuria; gout history</td><td align="left" valign="top">2</td><td align="left" valign="top">Case ID: C05, C28</td></tr><tr><td align="left" valign="top" colspan="4">Mixed and complex abnormalities (n=10)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x2003;Mixed CaOx + uric acid</named-content></td><td align="left" valign="top">Hypercalciuria combined with hyperuricosuria and/or low pH</td><td align="left" valign="top">3</td><td align="left" valign="top">Case ID: C07, C20</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Infection stones (struvite)</td><td align="left" valign="top">High urine pH (&#x003E;7.5); high ammonium; recurrent UTIs<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">2</td><td align="left" valign="top">Case ID: C09</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Enteric hyperoxaluria</td><td align="left" valign="top">Severe hyperoxaluria (&#x003E;1.0 mmol/d) due to malabsorption (eg, Crohn, IBD<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>)</td><td align="left" valign="top">2</td><td align="left" valign="top">Case ID: C08</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Renal tubular acidosis</td><td align="left" valign="top">High urine pH (&#x003E;6.8); severe hypocitraturia; calcium phosphate stones</td><td align="left" valign="top">2</td><td align="left" valign="top">Case ID: C10, C18</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cystinuria</td><td align="left" valign="top">Genetic defect; positive urinary cystine</td><td align="left" valign="top">1</td><td align="left" valign="top">Case ID: C11</td></tr><tr><td align="left" valign="top" colspan="4">Safety-critical scenarios (n=8)</td></tr><tr><td align="left" valign="top">Congestive heart failure</td><td align="left" valign="top">Fluid volume sensitivity; potential harm from sodium bicarbonate load</td><td align="left" valign="top">2</td><td align="left" valign="top">Case ID: C21</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Chronic kidney disease</td><td align="left" valign="top">Stage 3b-4 (GFR<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup>&#x003C;45); risk of hyperkalemia and fluid overload</td><td align="left" valign="top">2</td><td align="left" valign="top">Case ID: C22</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pregnancy</td><td align="left" valign="top">Physiological hypercalciuria; restrictions on pharmacotherapy</td><td align="left" valign="top">2</td><td align="left" valign="top">Case ID: C23</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Primary hyperparathyroidism</td><td align="left" valign="top">Hypercalcemia; surgical indication rather than dietary alone</td><td align="left" valign="top">1</td><td align="left" valign="top">Case ID: C16</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Urinary incontinence</td><td align="left" valign="top">Older adults; risk of worsening symptoms with aggressive hydration</td><td align="left" valign="top">1</td><td align="left" valign="top">Case ID: C24</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>The number of representative cases listed for each phenotype does not always equal the total case count because only selected examples are provided to illustrate the clinical category. All 30 synthetic clinical vignettes were included in the evaluation dataset.</p></fn><fn id="table1fn2"><p><sup>b</sup>UTI: urinary tract infection.</p></fn><fn id="table1fn3"><p><sup>c</sup>IBD: inflammatory bowel disease.</p></fn><fn id="table1fn4"><p><sup>d</sup>GFR: glomerular filtration rate.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-1-5"><title>Reference Standard</title><p>For each vignette, a guideline-concordant dietary plan was established as the reference standard by 2 senior endourologists (&#x003E;15 y of experience in stone management) using EAU and CUA recommendations [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Discrepancies were resolved by consensus. The complete set of reference standard recommendations is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec></sec><sec id="s2-2"><title>Development of the StoneAgent Framework</title><sec id="s2-2-1"><title>Underlying Model and Configuration</title><p>The recommendations evaluated in this study were generated using an LLM accessed through the GPT-5.1 web interface. The experiments were conducted between December 2025 and January 2026 using the model version available on the platform at that time. Both the StoneAgent framework and the standard LLM condition used the same underlying model; the only difference between conditions was the interaction framework and the prompt structure applied to the model.</p><p>Because the web interface does not provide direct control over generation parameters such as temperature, top-p sampling, or fixed model snapshot identifiers, all outputs were produced using the default platform configuration. Accordingly, this study should be interpreted as an operational comparative evaluation under real-world usage conditions rather than a fully parameter-controlled API benchmark. To minimize context carryover, each clinical vignette or query variant was evaluated in a separate, newly initiated conversation session. No custom instructions, memory functions, or external tools were used during generation.</p><p>Two framework conditions were evaluated. The first, StoneAgent, represented a guideline-integrated prompting framework designed to support the interpretation of metabolic evaluation results and the generation of personalized dietary recommendations. The second represented a standard, general-purpose LLM condition, in which the same clinical content was presented without the structured clinical reasoning scaffold. To enhance transparency and reproducibility, the full prompt templates were prespecified and are reported in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendices 3</xref> and <xref ref-type="supplementary-material" rid="app4">4</xref>, and the complete model outputs were retained and shared in the supplementary materials.</p></sec><sec id="s2-2-2"><title>Prompt Engineering Strategy</title><p>The StoneAgent framework was developed to address the specific requirements of metabolic stone prevention counseling rather than to improve general conversational performance. The framework design was informed by the clinical workflow used by specialists when interpreting 24-hour urine metabolic evaluations and translating findings into dietary recommendations. Specifically, the prompt structure was organized into 5 sequential components: (1) clinical role orientation, which established a specialist perspective for metabolic stone management; (2) task definition, which instructed the model to interpret the clinical scenario and generate personalized dietary guidance; (3) metabolic reasoning structure, which required identification of dominant urinary abnormalities and their potential dietary drivers; (4) safety constraints, which prompted consideration of comorbid conditions and situations requiring modification of standard advice; and (5) output organization, which emphasized prioritized, patient-usable recommendations. This structure was designed to mirror key steps in clinical dietary counseling while remaining compatible with a general-purpose LLM.</p><p>StoneAgent was designed as a guideline-integrated response framework for the interpretation of metabolic stone evaluations. Rather than relying on the model&#x2019;s default conversational behavior, the framework imposed a structured clinical task orientation: identifying the dominant metabolic abnormalities, linking those abnormalities to likely dietary drivers, prioritizing preventive recommendations, and checking whether standard advice required modification because of comorbid conditions or other safety constraints. The output was structured to emphasize patient-usable dietary counseling while preserving fidelity to established clinical guidance.</p><p>Within this framework, the model was instructed to interpret the clinical vignette from the perspective of a clinician familiar with metabolic stone disease and current guideline recommendations. The instructions emphasized alignment with the principles outlined in the EAU and CUA guidelines [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>] and highlighted the importance of considering patient safety when comorbid conditions were present.</p><p>The prompt structure also encouraged the model to review the metabolic profile before generating recommendations. Urinary abnormalities, such as hypercalciuria, hypocitraturia, hyperoxaluria, and low urine volume, were expected to be interpreted in relation to established metabolic risk patterns.</p><p>The generated output focused on practical dietary guidance that could reasonably be communicated to patients during clinical counseling. Recommendations were, therefore, constrained to clear and actionable lifestyle measures rather than broad or nonspecific advice.</p><p>The complete prompt templates used in the StoneAgent configuration, together with the full set of model-generated responses for all clinical vignettes, are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p></sec><sec id="s2-2-3"><title>Agent Reasoning Framework</title><p>The StoneAgent framework was designed to approximate the reasoning process commonly used by clinicians when interpreting metabolic evaluations for kidney stone prevention. When a clinical vignette was provided as input, the model first examined the clinical context, including patient demographics, comorbidities, and prior stone history. These elements were considered to ensure that any dietary recommendations would remain appropriate for the patient&#x2019;s overall clinical condition.</p><p>The model then analyzed the 24-hour urine metabolic profile to identify abnormalities associated with stone formation. Identified abnormalities were subsequently interpreted in relation to potential dietary or metabolic etiologies, allowing the system to link laboratory findings with modifiable lifestyle factors relevant to stone prevention. Parameters such as urinary calcium, citrate, oxalate, sodium, and urinary pH were interpreted in relation to established metabolic risk patterns described in clinical guidelines [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>Based on this interpretation, the model generated dietary recommendations tailored to the metabolic findings and clinical context. The output focused on modifiable dietary and lifestyle factors commonly addressed in stone prevention, including fluid intake, sodium restriction, calcium consumption, oxalate exposure, and citrate supplementation, where appropriate.</p><p>This reasoning structure was intended to reflect guideline-based clinical decision-making and to reduce the likelihood that the model would generate generic or inconsistent recommendations.</p></sec></sec><sec id="s2-3"><title>Experimental Setup and Comparison Groups</title><sec id="s2-3-1"><title>Study Conditions and Reference Standard</title><p>To evaluate the contribution of structured clinical framing rather than differences in underlying model capability, we compared 2 interaction frameworks implemented on the same base LLM against a vignette-specific expert reference standard. The 2 conditions intentionally used different prompt structures because the objective was to evaluate whether a guideline-integrated clinical interaction framework could improve the reliability and usefulness of a general-purpose LLM for a specialty counseling task. In the StoneAgent condition, the model received the vignette within the guideline-integrated framework described above, which incorporated structured metabolic interpretation, safety considerations, and patient-oriented output organization. In the standard LLM condition, the same clinical content was presented without the additional reasoning scaffold, allowing the model to respond using its default general-purpose conversational behavior. Therefore, the observed differences should be interpreted as the contribution of framework-level design elements, including clinical framing, guideline grounding, and safety-oriented scaffolding, rather than as differences in the intrinsic capability of the underlying language model.</p><p>The expert reference standard was developed for each vignette by 2 senior endourologists with experience in metabolic stone prevention. Each expert independently reviewed the vignette and drafted guideline-concordant dietary recommendations based on current EAU and CUA guidance [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Differences were resolved through discussion to create a single consensus reference response for scoring. Full prompt templates and complete model outputs for the structured-input and patient query&#x2013;style robustness experiments are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendices 3</xref> and <xref ref-type="supplementary-material" rid="app4">4</xref>.</p></sec><sec id="s2-3-2"><title>Query-Style Input Robustness Experiment</title><p>To assess robustness under more naturalistic input conditions, each structured vignette was reformulated into 2 patient query&#x2013;style variants (query A and query B). These variants preserved the same underlying clinical scenario and 24-hour urine findings, but presented the information in a conversational format with realistic differences in wording, emphasis, and information order. No new clinical facts were introduced beyond those contained in the original vignette.</p><p>For each query variant, StoneAgent and the standard LLM configuration generated responses in separate new conversation sessions to minimize carryover effects from prior prompts or outputs. This design allowed us to examine whether the relative performance of the 2 systems was preserved when the same clinical content was expressed in a more patient-like manner. Because query A and query B represented 2 phrasings of the same clinical case, rather than independent cases, the primary robustness analysis was performed at the vignette level after aggregating the 2 query variants for each case.</p></sec></sec><sec id="s2-4"><title>Outcome Measures and Evaluation Procedure</title><p>Outputs from the 2 model conditions were evaluated using a structured scoring framework anchored to the vignette-specific expert reference standard. For each case, scoring focused on whether the response correctly identified the dominant metabolic pattern, prioritized recommendations consistent with guideline-based care, translated the findings into practical patient-facing advice, and avoided clinically inappropriate suggestions.</p><p>Three independent reviewers, blinded to study arm allocation, evaluated all outputs using the same predefined rubric. To reduce the likelihood of recognition bias, outputs were deidentified and scored according to content rather than response source. The evaluation framework was applied to both the structured-input experiment and the patient query&#x2013;style robustness experiment. The detailed scoring rubric is summarized in <xref ref-type="table" rid="table2">Table 2</xref>, and the final reviewer-scoring dataset is provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Standardized scoring rubric used by expert reviewers to evaluate AI responses.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Score</td><td align="left" valign="bottom">Metabolic specificity</td><td align="left" valign="bottom">Guideline adherence</td><td align="left" valign="bottom">Actionability</td></tr></thead><tbody><tr><td align="left" valign="top">1 (Very poor)</td><td align="left" valign="top">Fails to identify relevant metabolic abnormalities; provides inaccurate or irrelevant recommendations</td><td align="left" valign="top">Direct contradiction of established guideline recommendations</td><td align="left" valign="top">Recommendations are unclear, impractical, or potentially misleading</td></tr><tr><td align="left" valign="top">2 (Poor)</td><td align="left" valign="top">Identifies abnormalities but does not link them to specific etiologies or provides largely generic advice</td><td align="left" valign="top">Partially inconsistent with guidelines or omits key management targets</td><td align="left" valign="top">Advice lacks clarity or clinical applicability</td></tr><tr><td align="left" valign="top">3 (Fair)</td><td align="left" valign="top">Correctly identifies major abnormalities but provides limited explanation of their clinical implications</td><td align="left" valign="top">Generally consistent with guidelines but lacks specificity or nuance</td><td align="left" valign="top">Recommendations are understandable but insufficiently prioritized</td></tr><tr><td align="left" valign="top">4 (Good)</td><td align="left" valign="top">Identifies specific metabolic drivers and links them to appropriate dietary modifications</td><td align="left" valign="top">Consistent with guideline-based targets and appropriate for the clinical context</td><td align="left" valign="top">Recommendations are clear, logically organized, and clinically useful</td></tr><tr><td align="left" valign="top">5 (Excellent)</td><td align="left" valign="top">Demonstrates comprehensive interpretation of metabolic findings, including interaction among abnormalities when present</td><td align="left" valign="top">Fully concordant with guideline recommendations, including context-specific considerations</td><td align="left" valign="top">Recommendations are clear, prioritized according to clinical importance, and readily applicable in practice</td></tr></tbody></table></table-wrap><p>Three ordinal domains were assessed: metabolic specificity, guideline adherence, and actionability. Each domain was scored on a 5-point Likert scale, with higher scores indicating better performance. For each output, scores from the 3 reviewers were averaged to generate a composite score for each ordinal domain.</p><p>Metabolic specificity reflected the extent to which the response was tailored to the metabolic abnormalities and the clinical scenario presented in the case. Higher scores were assigned when the output correctly prioritized the dominant stone-risk pattern and linked specific abnormalities, such as hypercalciuria, hypocitraturia, hyperoxaluria, low urine volume, or urine pH abnormalities, to appropriate dietary recommendations.</p><p>Guideline adherence was evaluated by the degree to which the recommendations were consistent with the expert reference standard and current guideline principles. Higher scores were awarded when the response captured the key recommended measures without introducing advice that conflicted with established dietary management strategies for recurrent stone prevention.</p><p>Actionability assessed whether responses translated the metabolic interpretation into clear, feasible, and patient-usable dietary guidance. Responses received higher scores when they provided specific instructions, practical priorities, and language that could reasonably support clinical counseling or patient follow-up.</p><p>In addition to these ordinal domains, safety was assessed as a binary outcome. A response was classified as unsafe if it omitted or contradicted an important case-specific safety constraint, particularly in scenarios in which otherwise standard dietary advice required modification because of comorbid disease, pregnancy, advanced age, infection-related stones, or other clinically relevant conditions. For the primary safety analysis, consensus failure was defined a priori as a fail judgment assigned by at least 2 of the 3 reviewers.</p></sec><sec id="s2-5"><title>Statistical Analysis</title><p>The primary unit of analysis was the clinical vignette. For the structured-input experiment, paired comparisons were performed between StoneAgent and the standard LLM for each vignette. For the patient query&#x2013;style robustness experiment, query A and query B represented 2 phrasings of the same underlying case rather than independent observations; therefore, reviewer-averaged scores for the 2 query variants were first aggregated at the vignette level before paired comparison.</p><p>For the ordinal domains of metabolic specificity, guideline adherence, and actionability, scores from the 3 independent reviewers were averaged to obtain a composite score for each output. These ordinal outcomes are summarized as medians with IQRs, and means are also reported for descriptive completeness where appropriate. Paired comparisons between StoneAgent and the standard LLM were performed using the Wilcoxon signed-rank test.</p><p>Safety was analyzed as a paired binary outcome. For each output, consensus failure was defined a priori as a fail judgment assigned by at least 2 of the 3 reviewers. Paired comparisons of safety pass rates between study arms were evaluated using the exact McNemar test.</p><p>Query-level results are presented descriptively to illustrate within-case consistency across the 2 patient-style phrasings, whereas vignette-level aggregation was used for the primary robustness comparison. Interrater reliability for the ordinal rubric domains was assessed using the intraclass correlation coefficient (ICC), based on a 2-way mixed-effects model for absolute agreement [<xref ref-type="bibr" rid="ref18">18</xref>]. Because reviewer scores were averaged to generate composite vignette-level scores, average-measures ICCs were used for the primary interpretation, with single-measure ICCs calculated for reference. Agreement for the binary safety assessment was summarized, where reviewer-level data were available, using percent agreement and Fleiss &#x03BA;. All tests were 2-sided, and <italic>P</italic>&#x003C;.05 was considered statistically significant. Given the exploratory nature of this in silico comparative study, secondary domain-level comparisons were interpreted as supportive rather than strictly confirmatory. All analyses were performed using SPSS, version 29.0 (IBM Corp).</p></sec><sec id="s2-6"><title>Reporting Guideline</title><p>This study was reported with consideration of the relevant recommendations for evaluating AI-based clinical interaction and decision support approaches. The completed checklist is provided as a supplementary file.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overall Performance</title><p>A total of 30 synthetic clinical vignettes were evaluated in the structured-input experiment, and each vignette was also reformulated into 2 patient query&#x2013;style variants for the robustness experiment. Across both phases, StoneAgent generally outperformed the standard LLM in metabolic specificity, guideline adherence, and actionability, while also demonstrating a more favorable safety profile. The full prompt and response sets are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendices 3</xref> and <xref ref-type="supplementary-material" rid="app4">4</xref>, and the reviewer scoring dataset is provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p><p>In the structured-input experiment, StoneAgent showed consistently higher paired ratings than the standard LLM across all 3 ordinal domains. Median scores reached 5.00 across metabolic specificity, guideline adherence, and actionability, whereas the standard LLM showed lower and more variable performance. Safety pass rates were numerically higher for StoneAgent (30/30 vs 25/30), although this difference did not reach statistical significance in exact paired testing.</p><p>In the patient query&#x2013;style experiment, the absolute differences between frameworks were more conservative after vignette-level aggregation of query A and query B, but the directional advantage of StoneAgent was retained across the 3 ordinal domains, with numerically higher safety pass rates as well. Taken together, these findings indicate that the observed advantage of StoneAgent was not limited to idealized structured inputs and remained detectable when the same cases were presented in more natural conversational language.</p></sec><sec id="s3-2"><title>Interrater Reliability</title><p>Interrater reliability across the 3 ordinal rubric domains was excellent in both experimental phases. In the structured-input experiment, the average-measures ICC(A,3) was 0.991 for metabolic specificity, 0.961 for guideline adherence, and 0.967 for actionability. In the patient query&#x2013;style robustness experiment, after vignette-level aggregation of query A and query B, ICC(A,3) values were 0.978 for metabolic specificity, 0.988 for guideline adherence, and 0.974 for actionability. These findings support the use of reviewer-averaged composite scores for vignette-level comparison.</p><p>For the binary safety assessment, reviewer-level ratings were available in the query-level dataset, where agreement was also high (Fleiss &#x03BA;=0.944; overall agreement 99.4%). For the structured-input experiment, only the final consensus safety classification was retained in the exported appendix workbook.</p><p>A summary of the paired case-level results for the structured-input experiment is presented in <xref ref-type="table" rid="table3">Table 3</xref>.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Case-level performance of StoneAgent vs a standard large language model (LLM) in the structured-input experiment<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Domain</td><td align="left" valign="bottom">StoneAgent</td><td align="left" valign="bottom">Standard LLM</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Metabolic specificity, median (IQR)</td><td align="left" valign="top">5.00 (5.00&#x2010;5.00)</td><td align="left" valign="top">3.00 (2.75&#x2010;3.92)</td><td align="char" char="." valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Guideline adherence, median (IQR)</td><td align="left" valign="top">5.00 (5.00&#x2010;5.00)</td><td align="left" valign="top">3.67 (3.08&#x2010;4.00)</td><td align="char" char="." valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Actionability, median (IQR)</td><td align="left" valign="top">5.00 (5.00&#x2010;5.00)</td><td align="left" valign="top">3.00 (2.67&#x2010;3.33)</td><td align="char" char="." valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Clinical safety, n/N (%)</td><td align="left" valign="top">30/30 (100)</td><td align="left" valign="top">25/30 (83.3)</td><td align="left" valign="top">.06 (exact McNemar)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Values are presented as median (IQR). Paired comparisons for ordinal domains were performed using the Wilcoxon signed-rank test. Safety outcomes were compared using the exact McNemar test.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Metabolic Specificity</title><sec id="s3-3-1"><title>Quantitative and Qualitative Findings</title><p>In the structured-input experiment, StoneAgent received higher metabolic specificity scores than the standard LLM, indicating more consistent recognition and prioritization of the dominant metabolic abnormalities within each vignette (median 5.00, IQR 5.00&#x2010;5.00 vs median 3.00, IQR 2.75&#x2010;3.92; Wilcoxon signed-rank <italic>P</italic>&#x003C;.001). In the patient query&#x2013;style robustness experiment, this advantage was retained after aggregation of query A and query B at the vignette level (median 4.67, IQR 4.00&#x2010;5.00 vs median 4.00, IQR 3.00&#x2010;4.00; <italic>P</italic>&#x003C;.001), suggesting that StoneAgent remained better able to map patient-style input back to the underlying metabolic-risk pattern.</p><p>Qualitatively, StoneAgent&#x2019;s responses more often linked specific abnormalities, such as hypercalciuria, hyperoxaluria, hypocitraturia, urine pH abnormalities, or low urine volume, to corresponding dietary targets. By contrast, the standard LLM more often defaulted to broadly appropriate but less case-prioritized advice, particularly in scenarios requiring identification of a dominant metabolic driver rather than generic stone-prevention counseling.</p><p>The tabulated results in <xref ref-type="table" rid="table3">Tables 3</xref> and <xref ref-type="table" rid="table4">4</xref> are complemented by the score distributions shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Case-level performance of StoneAgent vs a standard large language model (LLM) in the patient query&#x2013;style robustness experiment<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Domain</td><td align="left" valign="bottom">StoneAgent</td><td align="left" valign="bottom">Standard LLM</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Mean difference</td></tr></thead><tbody><tr><td align="left" valign="top">Metabolic specificity, median (IQR)</td><td align="left" valign="top">4.67 (4.00&#x2010;5.00)</td><td align="left" valign="top">4.00 (3.00&#x2010;4.00)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">+0.82</td></tr><tr><td align="left" valign="top">Guideline adherence, median (IQR)</td><td align="left" valign="top">5.00 (4.00&#x2010;5.00)</td><td align="left" valign="top">4.00 (3.00&#x2010;4.00)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">+0.88</td></tr><tr><td align="left" valign="top">Actionability, median (IQR)</td><td align="left" valign="top">4.00 (4.00&#x2010;4.00)</td><td align="left" valign="top">3.00 (3.00&#x2010;4.00)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">+0.71</td></tr><tr><td align="left" valign="top">Clinical safety, n/N (%)</td><td align="left" valign="top">30/30 (100)</td><td align="left" valign="top">27/30 (90)</td><td align="left" valign="top">.25 (exact McNemar)</td><td align="left" valign="top">+10 percentage points</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Ordinal domain scores represent vignette-level composite ratings from 3 blinded reviewers after aggregation of query A and query B. Paired comparisons were performed using the Wilcoxon signed-rank test and are summarized as medians (IQRs); mean differences are shown for descriptive comparison. Clinical safety was analyzed separately as a binary outcome using the exact McNemar test.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Distribution of case-level evaluation scores and safety outcomes for StoneAgent vs a standard large language model (LLM) across the structured-input and patient query&#x2013;style experiments. Panels A and B show the distributions of case-level mean scores for metabolic specificity, guideline adherence, and actionability. Each point represents 1 vignette; boxplots indicate the median and IQR. For the patient query&#x2013;style robustness experiment, query A and query B were aggregated at the vignette level before analysis. Panel C shows safety pass rates for each model condition in both experimental phases. <italic>P</italic> values for ordinal domains were calculated using the Wilcoxon signed-rank test, and safety was compared using the exact McNemar test.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95162_fig02.png"/></fig></sec><sec id="s3-3-2"><title>Guideline Adherence</title><p>StoneAgent also showed higher guideline adherence than the standard LLM in the structured-input experiment (median 5.00, IQR 5.00&#x2010;5.00 vs median 3.67, IQR 3.08&#x2010;4.00; <italic>P</italic>&#x003C;.001). In the patient query&#x2013;style robustness experiment, the same directional advantage remained present after vignette-level aggregation (median 5.00, IQR 4.00&#x2010;5.00 vs median 4.00, IQR 3.00&#x2010;4.00; <italic>P</italic>&#x003C;.001).</p><p>The difference was most apparent in cases in which guideline-concordant counseling required prioritization rather than simple completeness. These included scenarios in which standard stone-prevention advice needed to be modified because of infection-related stones, primary hyperparathyroidism, chronic kidney disease, heart failure, pregnancy, or other clinically relevant constraints. In such cases, StoneAgent more often preserved the central management priority and avoided advice that was superficially plausible but incompletely aligned with the reference standard.</p></sec><sec id="s3-3-3"><title>Actionability</title><p>StoneAgent responses were also rated more actionable than those generated by the standard LLM. In the structured-input experiment, StoneAgent more often translated metabolic findings into clear, patient-usable priorities, such as sodium restriction targets, appropriate calcium intake strategies, fluid goals, oxalate-related counseling, or citrate-focused dietary measures (median 5.00, IQR 5.00&#x2010;5.00 vs median 3.00, IQR 2.67&#x2010;3.33; <italic>P</italic>&#x003C;.001). In the patient query&#x2013;style experiment, this advantage remained present, although the difference was more modest than that observed under structured inputs (median 4.00, IQR 4.00&#x2010;4.00 vs median 3.00, IQR 3.00&#x2010;4.00; <italic>P</italic>&#x003C;.001).</p><p>The reported scores for metabolic specificity, guideline adherence, and actionability in the patient query&#x2013;style robustness experiment represent vignette-level composite scores derived from ratings by 3 blinded reviewers. For this phase, each vignette was reformulated into 2 patient-style prompts (query A and query B), and reviewer-averaged scores were aggregated at the vignette level to avoid treating 2 phrasings of the same case as independent observations. Safety was analyzed separately as a binary outcome. A summary of the aggregated patient query&#x2013;style comparison is presented in <xref ref-type="table" rid="table4">Table 4</xref>.</p></sec></sec><sec id="s3-4"><title>Clinical Safety</title><p>StoneAgent showed a numerically more favorable safety profile than the standard LLM across both experimental phases (<xref ref-type="table" rid="table3">Tables 3</xref> and <xref ref-type="table" rid="table4">4</xref>). In the structured-input experiment, StoneAgent produced no consensus-fail responses, whereas 5 of 30 responses in the standard LLM group were classified as unsafe or insufficiently constrained (30/30 vs 25/30; exact McNemar <italic>P</italic>=.06). In the patient query&#x2013;style robustness experiment, the same directional pattern persisted after vignette-level aggregation (30/30 vs 27/30; exact McNemar <italic>P</italic>=.25). These findings suggest a potential safety advantage associated with the StoneAgent framework, although the paired comparisons for safety did not reach conventional statistical significance in this sample.</p><p>The safety difference was most evident in vignettes in which standard stone-prevention recommendations required explicit modification because of comorbid conditions or competing clinical priorities. These included cases involving advanced chronic kidney disease, heart failure, pregnancy, primary hyperparathyroidism, infection-related stones, and older adults with functional constraints affecting hydration counseling.</p></sec><sec id="s3-5"><title>Illustrative Cases</title><p>Representative cases illustrated how framework-level differences translated into reviewer scoring. In case C09, an infection-related stone scenario, StoneAgent more consistently prioritized infection-focused management and stone clearance before routine dietary prevention advice, whereas the standard LLM was more likely to default to generic recurrence-prevention counseling [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. In case C22, a chronic kidney disease scenario, StoneAgent more often modified otherwise standard recommendations related to fluid intake, alkalinization, or electrolyte-related advice in light of renal safety considerations, whereas the standard LLM was more prone to provide broadly reasonable but insufficiently constrained recommendations [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. These examples highlight that the observed differences were not limited to response detail alone but also involved prioritization of the dominant clinical problem and preservation of case-specific safety logic.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study evaluated whether a guideline-integrated, safety-aware clinical interaction framework could improve the quality of personalized dietary counseling outputs for recurrent urolithiasis when implemented on the same underlying LLM as the standard LLM interaction condition. In the structured-input experiment, StoneAgent generated recommendations that were more metabolically specific, more closely aligned with guideline-based management, and more actionable than those produced by the standard LLM configuration. Importantly, this performance advantage was not confined to standardized vignette inputs. When the same cases were reformulated as patient query&#x2013;style prompts, StoneAgent retained a directional advantage across the main evaluation domains after vignette-level aggregation.</p><p>The observed difference was not simply a matter of producing longer or more detailed answers. Rather, StoneAgent more consistently identified the dominant metabolic problem, prioritized the most relevant preventive strategy, and modified standard dietary advice when case-specific safety constraints were present. The contrast with the standard LLM was most evident in vignettes involving comorbidity-sensitive counseling, mixed metabolic patterns, or scenarios in which broadly reasonable stone advice could still be clinically misprioritized.</p><p>Taken together, these findings support the view that specialty counseling tasks may benefit not only from general language competence but also from explicit clinical framing, guideline grounding, and safety-oriented response scaffolding.</p><p>It is important to interpret guideline adherence in the context of the framework design. Because explicit guideline integration was a predefined component of the StoneAgent framework, this outcome reflects the degree to which the framework successfully operationalized guideline-based counseling principles rather than representing an isolated measure of general reasoning ability. Therefore, the comparison evaluates the clinical use of a structured, guideline-oriented interaction approach relative to an unguided LLM interaction.</p></sec><sec id="s4-2"><title>Comparison With Previous Work</title><p>Prior studies on LLM use in urology have largely focused on general health information, patient education, or broad question-answering tasks [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref23">23</xref>]. By contrast, the present study addressed a narrower and more clinically constrained task: translating 24-hour urine metabolic findings into personalized dietary recommendations for recurrent stone prevention. This distinction is important because the quality of the response in this setting depends not only on fluency or factual recall but also on correctly prioritizing metabolic abnormalities, preserving guideline-consistent management logic, and respecting case-specific safety constraints.</p><p>Our findings also extend the literature by evaluating performance under both structured vignette inputs and patient query&#x2013;style inputs. This dual design allowed us to assess not only whether a framework could perform well under standardized conditions but also whether its advantage was retained when the same cases were expressed in more natural conversational language. In that sense, the study evaluates whether a clinically structured, guideline-integrated interaction framework can improve task reliability across different input formats when compared with a standard LLM interaction.</p><p>More broadly, as general-purpose foundation models continue to improve, the key question may shift from whether they can respond to specialized counseling tasks at all to how they should be structured and governed for reliable use. Our findings suggest that explicit clinical framing and safety-aware scaffolding may still add value even when a general-purpose model is already capable of producing broadly plausible responses [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>].</p></sec><sec id="s4-3"><title>Clinical Implications</title><p>These findings have potential implications for metabolic stone counseling, particularly in settings where clinicians must convert laboratory data into patient-facing advice under time constraints. A framework such as StoneAgent may support dietary counseling by organizing metabolic interpretation into more consistent and practical recommendations. Its potential value may be greatest not in replacing specialist judgment but in assisting with the translation of metabolic findings into structured counseling points that are easier to communicate during follow-up visits or preventive care discussions.</p><p>From an implementation perspective, the potential value of such a framework will depend not only on clinical performance but also on development, maintenance, and integration costs. This study did not perform a health economic evaluation and therefore cannot determine whether framework-assisted counseling would reduce health care expenditures. However, recurrent urolithiasis is associated with substantial downstream costs related to repeated procedures, emergency visits, and long-term management. Future studies should evaluate the cost-effectiveness of guideline-integrated LLM-based counseling tools by considering implementation costs alongside potential reductions in preventable recurrence-related health care service use.</p><p>However, higher actionability scores should not necessarily be interpreted as requiring increasingly detailed or prescriptive recommendations. In dietary counseling for recurrent urolithiasis, clinically useful advice must balance specificity with appropriate individualization and safety considerations. Particularly in cases involving comorbidities or competing clinical priorities, a cautious and appropriately constrained recommendation may be preferable to a more extensive but potentially unsuitable dietary plan.</p><p>At the same time, the results should not be interpreted as support for fully autonomous dietary management. The outputs evaluated in this study were generated in a simulated setting and should be viewed as decision support candidates rather than stand-alone clinical advice.</p></sec><sec id="s4-4"><title>Safety Considerations</title><p>Safety is a particularly important dimension in dietary counseling for recurrent urolithiasis because recommendations that appear broadly reasonable in general stone prevention may become inappropriate when important clinical modifiers are present, such as chronic kidney disease, pregnancy, infection-related stones, advanced age, or cardiovascular comorbidity. In this context, the apparent advantage of StoneAgent is less about generating longer or more detailed responses and more about preserving case-specific clinical constraints when translating metabolic findings into advice. Although the safety comparisons in this study did not reach conventional statistical significance, the directional pattern across both experimental phases suggests that explicit safety-oriented scaffolding may be valuable in specialty counseling tasks where otherwise standard recommendations require contextual modification [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref26">26</xref>].</p><p>The potential risk of unguided LLM use should also be considered. Although general-purpose LLMs can provide broadly appropriate health information, responses generated without clinical constraints may fail to prioritize the dominant metabolic abnormality, provide overly generalized dietary advice, or recommend modifications that are inappropriate for patients with specific comorbidities. In recurrent stone disease, such limitations could theoretically contribute to ineffective prevention strategies, unnecessary dietary restriction, or delayed recognition of conditions requiring specialist evaluation. Therefore, future clinical applications of LLMs in specialty counseling should incorporate explicit safety frameworks, domain-specific validation, and appropriate human oversight.</p></sec><sec id="s4-5"><title>Limitations</title><p>Several limitations should be considered. First, this study was based on synthetic clinical vignettes rather than real patient encounters. Although the cases were designed to reflect clinically plausible and guideline-relevant metabolic stone scenarios, they cannot capture the full heterogeneity, ambiguity, and communication variability of real-world practice. Second, the patient query&#x2013;style inputs were derived from the structured vignettes rather than authored by patients themselves and therefore represent a pragmatic approximation of real consultation language rather than a true external validation set. Third, the study was designed as an exploratory in silico benchmarking study using purposively constructed cases to maximize coverage of guideline-relevant and safety-sensitive scenarios, rather than as a power-calculated confirmatory trial. Fourth, the study compared 2 interaction frameworks implemented through the same publicly available web interface of the underlying LLM. Because that interface does not provide API-level control over generation parameters or fixed model snapshot identifiers, the findings should be interpreted as an operational comparative evaluation under real-world usage conditions rather than as a fully parameter-controlled benchmark. Fifth, the study focused specifically on dietary counseling and did not evaluate broader management decisions such as pharmacologic prevention, imaging follow-up, or procedural planning. Finally, response quality was assessed against expert-derived standards rather than real patient outcomes, and the study evaluated isolated response generation rather than longitudinal patient-model interactions. Therefore, the findings do not establish clinical effectiveness, adherence, recurrence reduction, conversational adaptation, or long-term clinical integration.</p><p>Additionally, because the intervention being evaluated was a structured prompting framework, the observed performance improvement may partly reflect the contribution of prompt design and task-specific scaffolding rather than the language model alone. Future studies using controlled prompt-ablation designs could help clarify the relative contribution of individual framework components.</p></sec><sec id="s4-6"><title>Future Directions</title><p>Future work should evaluate this type of framework using real patient-authored queries, external clinician reviewers, and prospective clinical workflows. Future evaluations should also examine interactive use cases rather than single-turn responses alone, including how guideline-integrated frameworks respond to additional patient questions, clarification requests, newly available laboratory information, or changes in clinical context over time. Future evaluations should also include patients or simulated cases with multiple concurrent comorbidities to determine whether the framework can appropriately reconcile competing or interacting safety constraints when generating dietary recommendations. Future development could also distinguish between clinician-facing and patient-facing implementations of guideline-integrated LLM frameworks. A clinician-facing version could support interpretation of metabolic findings and preparation of individualized counseling plans within clinical workflows, with subsequent evaluation against patient outcomes. A patient-facing version could support longitudinal reinforcement of clinician-provided recommendations, respond to follow-up questions, and adapt counseling as symptoms, laboratory findings, or clinical circumstances change between formal health care encounters. These complementary use cases warrant separate evaluation of usability, safety, and clinical effectiveness. Qualitative studies involving patients and clinicians, such as interviews or focus groups with individuals who have experience with recurrent stone disease, may provide important insights into usability, trust, communication quality, and barriers to adoption. Comparative evaluation across newer foundation models and alternative guideline-based frameworks would also help clarify which components of the observed benefit are attributable to framework design and which may diminish as baseline model performance improves.</p></sec><sec id="s4-7"><title>Conclusions</title><p>In this in silico comparative study, a guideline-integrated, safety-aware clinical interaction framework generated higher-quality dietary recommendations for recurrent urolithiasis than a standard general-purpose LLM interaction when both were implemented on the same underlying model. This advantage was retained, although in a more conservative form, when the same cases were presented as patient query&#x2013;style inputs. These findings support the value of explicit clinical framing, guideline grounding, and safety-oriented scaffolding for specialty counseling tasks involving metabolic stone prevention. Further evaluation is needed using real patient-authored queries, external reviewers, and prospective clinical settings.</p></sec></sec></body><back><ack><p>The authors thank the senior urologist who reviewed the synthetic clinical vignettes for clinical plausibility and consistency with routine metabolic stone evaluation scenarios. Generative AI tools were not used for study design, data generation, data analysis, interpretation of results, or scientific decision-making. During manuscript revision, AI-assisted tools (ChatGPT; OpenAI) were used only to refine language and improve readability. All revisions were critically reviewed by the authors, who take full responsibility for the scientific content and the final manuscript.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec><sec><title>Data Availability</title><p>The data supporting the findings of this study are available from the corresponding author upon reasonable request. Supporting materials submitted with this manuscript include 5 multimedia appendices containing the synthetic vignette dataset, reference materials, prompts and model outputs, patient query&#x2013;style input variants, and reviewer scoring materials.</p></sec></notes><fn-group><fn fn-type="con"><p>XW and JL jointly conceived and designed the study. JY contributed to data review and data verification. All authors reviewed the final manuscript and approved the submitted version.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CUA</term><def><p>Canadian Urological Association</p></def></def-item><def-item><term id="abb2">EAU</term><def><p>European Association of Urology</p></def></def-item><def-item><term id="abb3">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sorokin</surname><given-names>I</given-names> </name><name name-style="western"><surname>Mamoulakis</surname><given-names>C</given-names> </name><name name-style="western"><surname>Miyazawa</surname><given-names>K</given-names> </name><name name-style="western"><surname>Rodgers</surname><given-names>A</given-names> </name><name name-style="western"><surname>Talati</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lotan</surname><given-names>Y</given-names> </name></person-group><article-title>Epidemiology of stone disease across the world</article-title><source>World J Urol</source><year>2017</year><month>09</month><volume>35</volume><issue>9</issue><fpage>1301</fpage><lpage>1320</lpage><pub-id pub-id-type="doi">10.1007/s00345-017-2008-6</pub-id><pub-id pub-id-type="medline">28213860</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>D&#x2019;Costa</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Pais</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Rule</surname><given-names>AD</given-names> </name></person-group><article-title>Leave no stone unturned: defining recurrence in kidney stone formers</article-title><source>Curr Opin Nephrol Hypertens</source><year>2019</year><month>03</month><volume>28</volume><issue>2</issue><fpage>148</fpage><lpage>153</lpage><pub-id pub-id-type="doi">10.1097/MNH.0000000000000478</pub-id><pub-id pub-id-type="medline">30531469</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raheem</surname><given-names>OA</given-names> </name><name name-style="western"><surname>Khandwala</surname><given-names>YS</given-names> </name><name name-style="western"><surname>Sur</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Ghani</surname><given-names>KR</given-names> </name><name name-style="western"><surname>Denstedt</surname><given-names>JD</given-names> </name></person-group><article-title>Burden of urolithiasis: trends in prevalence, treatments, and costs</article-title><source>Eur Urol Focus</source><year>2017</year><month>02</month><volume>3</volume><issue>1</issue><fpage>18</fpage><lpage>26</lpage><pub-id pub-id-type="doi">10.1016/j.euf.2017.04.001</pub-id><pub-id pub-id-type="medline">28720363</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patel</surname><given-names>N</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>RD</given-names> </name><name name-style="western"><surname>Sarkissian</surname><given-names>C</given-names> </name><name name-style="western"><surname>De</surname><given-names>S</given-names> </name><name name-style="western"><surname>Monga</surname><given-names>M</given-names> </name></person-group><article-title>Quality of life and urolithiasis: the patient-reported outcomes measurement information system (PROMIS)</article-title><source>Int Braz J Urol</source><year>2017</year><volume>43</volume><issue>5</issue><fpage>880</fpage><lpage>886</lpage><pub-id pub-id-type="doi">10.1590/S1677-5538.IBJU.2016.0649</pub-id><pub-id pub-id-type="medline">28792186</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johnston</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>BPH</given-names> </name><name name-style="western"><surname>Rai</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Incremental healthcare cost implications of retreatment following ureteroscopy or percutaneous nephrolithotomy for upper urinary tract stones: a population-based study of commercially-insured US adults</article-title><source>Med Devices (Auckl)</source><year>2022</year><volume>15</volume><fpage>371</fpage><lpage>384</lpage><pub-id pub-id-type="doi">10.2147/MDER.S384823</pub-id><pub-id pub-id-type="medline">36389203</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Skolarikos</surname><given-names>A</given-names> </name><name name-style="western"><surname>Somani</surname><given-names>B</given-names> </name><name name-style="western"><surname>Neisius</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Metabolic evaluation and recurrence prevention for urinary stone patients: an EAU guidelines update</article-title><source>Eur Urol</source><year>2024</year><month>10</month><volume>86</volume><issue>4</issue><fpage>343</fpage><lpage>363</lpage><pub-id pub-id-type="doi">10.1016/j.eururo.2024.05.029</pub-id><pub-id pub-id-type="medline">39069389</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhojani</surname><given-names>N</given-names> </name><name name-style="western"><surname>Bjazevic</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wallace</surname><given-names>B</given-names> </name><etal/></person-group><article-title>UPDATE&#x2014;Canadian Urological Association guideline: evaluation and medical management of kidney stones</article-title><source>Can Urol Assoc J</source><year>2022</year><month>06</month><volume>16</volume><issue>6</issue><fpage>175</fpage><lpage>188</lpage><pub-id pub-id-type="doi">10.5489/cuaj.7872</pub-id><pub-id pub-id-type="medline">35623003</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goldfarb</surname><given-names>DS</given-names> </name><name name-style="western"><surname>Arowojolu</surname><given-names>O</given-names> </name></person-group><article-title>Metabolic evaluation of first-time and recurrent stone formers</article-title><source>Urol Clin North Am</source><year>2013</year><month>02</month><volume>40</volume><issue>1</issue><fpage>13</fpage><lpage>20</lpage><pub-id pub-id-type="doi">10.1016/j.ucl.2012.09.007</pub-id><pub-id pub-id-type="medline">23177631</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferraro</surname><given-names>PM</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>EN</given-names> </name><name name-style="western"><surname>Curhan</surname><given-names>GC</given-names> </name></person-group><article-title>24-hour urinary chemistries and kidney stone risk</article-title><source>Am J Kidney Dis</source><year>2024</year><month>08</month><volume>84</volume><issue>2</issue><fpage>164</fpage><lpage>169</lpage><pub-id pub-id-type="doi">10.1053/j.ajkd.2024.02.010</pub-id><pub-id pub-id-type="medline">38583757</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hsi</surname><given-names>RS</given-names> </name><name name-style="western"><surname>Sanford</surname><given-names>T</given-names> </name><name name-style="western"><surname>Goldfarb</surname><given-names>DS</given-names> </name><name name-style="western"><surname>Stoller</surname><given-names>ML</given-names> </name></person-group><article-title>The role of the 24-hour urine collection in the prevention of kidney stone recurrence</article-title><source>J Urol</source><year>2017</year><month>04</month><volume>197</volume><issue>4</issue><fpage>1084</fpage><lpage>1089</lpage><pub-id pub-id-type="doi">10.1016/j.juro.2016.10.052</pub-id><pub-id pub-id-type="medline">27746283</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Milose</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Kaufman</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Hollenbeck</surname><given-names>BK</given-names> </name><name name-style="western"><surname>Wolf</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Hollingsworth</surname><given-names>JM</given-names> </name></person-group><article-title>Prevalence of 24-hour urine collection in high risk stone formers</article-title><source>J Urol</source><year>2014</year><month>02</month><volume>191</volume><issue>2</issue><fpage>376</fpage><lpage>380</lpage><pub-id pub-id-type="doi">10.1016/j.juro.2013.08.080</pub-id><pub-id pub-id-type="medline">24018242</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSJ</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aydin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Karabacak</surname><given-names>M</given-names> </name><name name-style="western"><surname>Vlachos</surname><given-names>V</given-names> </name><name name-style="western"><surname>Margetis</surname><given-names>K</given-names> </name></person-group><article-title>Large language models in patient education: a scoping review of applications in medicine</article-title><source>Front Med (Lausanne)</source><year>2024</year><volume>11</volume><fpage>1477898</fpage><pub-id pub-id-type="doi">10.3389/fmed.2024.1477898</pub-id><pub-id pub-id-type="medline">39534227</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bedi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Orr-Ewing</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Testing and evaluation of health care applications of large language models: a systematic review</article-title><source>JAMA</source><year>2025</year><month>01</month><day>28</day><volume>333</volume><issue>4</issue><fpage>319</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1001/jama.2024.21700</pub-id><pub-id pub-id-type="medline">39405325</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gupta</surname><given-names>R</given-names> </name><name name-style="western"><surname>Pedraza</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Gorin</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Tewari</surname><given-names>AK</given-names> </name></person-group><article-title>Defining the role of large language models in urologic care and research</article-title><source>Eur Urol Oncol</source><year>2024</year><month>02</month><volume>7</volume><issue>1</issue><fpage>1</fpage><lpage>13</lpage><pub-id pub-id-type="doi">10.1016/j.euo.2023.07.017</pub-id><pub-id pub-id-type="medline">37648630</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Davis</surname><given-names>R</given-names> </name><name name-style="western"><surname>Eppler</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ayo-Ajibola</surname><given-names>O</given-names> </name><etal/></person-group><article-title>Evaluating the effectiveness of artificial intelligence-powered large language models application in disseminating appropriate and readable health information in urology</article-title><source>J Urol</source><year>2023</year><month>10</month><volume>210</volume><issue>4</issue><fpage>688</fpage><lpage>694</lpage><pub-id pub-id-type="doi">10.1097/JU.0000000000003615</pub-id><pub-id pub-id-type="medline">37428117</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Skolarikos</surname><given-names>A</given-names> </name><name name-style="western"><surname>Straub</surname><given-names>M</given-names> </name><name name-style="western"><surname>Knoll</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Metabolic evaluation and recurrence prevention for urinary stone patients: EAU guidelines</article-title><source>Eur Urol</source><year>2015</year><month>04</month><volume>67</volume><issue>4</issue><fpage>750</fpage><lpage>763</lpage><pub-id pub-id-type="doi">10.1016/j.eururo.2014.10.029</pub-id><pub-id pub-id-type="medline">25454613</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koo</surname><given-names>TK</given-names> </name><name name-style="western"><surname>Li</surname><given-names>MY</given-names> </name></person-group><article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research</article-title><source>J Chiropr Med</source><year>2016</year><volume>15</volume><issue>2</issue><fpage>155</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id><pub-id pub-id-type="medline">27330520</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Siener</surname><given-names>R</given-names> </name></person-group><article-title>Nutrition and kidney stone disease</article-title><source>Nutrients</source><year>2021</year><month>06</month><day>3</day><volume>13</volume><issue>6</issue><fpage>1917</fpage><pub-id pub-id-type="doi">10.3390/nu13061917</pub-id><pub-id pub-id-type="medline">34204863</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peerapen</surname><given-names>P</given-names> </name><name name-style="western"><surname>Thongboonkerd</surname><given-names>V</given-names> </name></person-group><article-title>Kidney stone prevention</article-title><source>Adv Nutr</source><year>2023</year><month>05</month><volume>14</volume><issue>3</issue><fpage>555</fpage><lpage>569</lpage><pub-id pub-id-type="doi">10.1016/j.advnut.2023.03.002</pub-id><pub-id pub-id-type="medline">36906146</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eppler</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Ganjavi</surname><given-names>C</given-names> </name><name name-style="western"><surname>Knudsen</surname><given-names>JE</given-names> </name><etal/></person-group><article-title>Bridging the gap between urological research and patient understanding: the role of large language models in automated generation of layperson's summaries</article-title><source>Urol Pract</source><year>2023</year><month>09</month><volume>10</volume><issue>5</issue><fpage>436</fpage><lpage>443</lpage><pub-id pub-id-type="doi">10.1097/UPJ.0000000000000428</pub-id><pub-id pub-id-type="medline">37410015</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gabriel</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shafik</surname><given-names>L</given-names> </name><name name-style="western"><surname>Alanbuki</surname><given-names>A</given-names> </name><name name-style="western"><surname>Larner</surname><given-names>T</given-names> </name></person-group><article-title>The utility of the ChatGPT artificial intelligence tool for patient education and enquiry in robotic radical prostatectomy</article-title><source>Int Urol Nephrol</source><year>2023</year><month>11</month><volume>55</volume><issue>11</issue><fpage>2717</fpage><lpage>2732</lpage><pub-id pub-id-type="doi">10.1007/s11255-023-03729-4</pub-id><pub-id pub-id-type="medline">37528247</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pompili</surname><given-names>D</given-names> </name><name name-style="western"><surname>Richa</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Collins</surname><given-names>P</given-names> </name><name name-style="western"><surname>Richards</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hennessey</surname><given-names>DB</given-names> </name></person-group><article-title>Using artificial intelligence to generate medical literature for urology patients: a comparison of three different large language models</article-title><source>World J Urol</source><year>2024</year><month>07</month><day>29</day><volume>42</volume><issue>1</issue><fpage>455</fpage><pub-id pub-id-type="doi">10.1007/s00345-024-05146-3</pub-id><pub-id pub-id-type="medline">39073590</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gul</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Monga</surname><given-names>M</given-names> </name></person-group><article-title>Medical and dietary therapy for kidney stone prevention</article-title><source>Korean J Urol</source><year>2014</year><month>12</month><volume>55</volume><issue>12</issue><fpage>775</fpage><lpage>779</lpage><pub-id pub-id-type="doi">10.4111/kju.2014.55.12.775</pub-id><pub-id pub-id-type="medline">25512810</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Frassetto</surname><given-names>L</given-names> </name><name name-style="western"><surname>Kohlstadt</surname><given-names>I</given-names> </name></person-group><article-title>Treatment and prevention of kidney stones: an update</article-title><source>Am Fam Physician</source><year>2011</year><month>12</month><day>1</day><volume>84</volume><issue>11</issue><fpage>1234</fpage><lpage>1242</lpage><pub-id pub-id-type="medline">22150656</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Prezioso</surname><given-names>D</given-names> </name><name name-style="western"><surname>Strazzullo</surname><given-names>P</given-names> </name><name name-style="western"><surname>Lotti</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Dietary treatment of urinary risk factors for renal stone formation. A review of CLU Working Group</article-title><source>Arch Ital Urol Androl</source><year>2015</year><month>07</month><day>7</day><volume>87</volume><issue>2</issue><fpage>105</fpage><lpage>120</lpage><pub-id pub-id-type="doi">10.4081/aiua.2015.2.105</pub-id><pub-id pub-id-type="medline">26150027</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Synthetic clinical vignette dataset.</p><media xlink:href="jmir_v28i1e95162_app1.xlsx" xlink:title="XLSX File, 17 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Expert reference standards for the synthetic clinical vignettes.</p><media xlink:href="jmir_v28i1e95162_app2.docx" xlink:title="DOCX File, 23 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Full prompts and model outputs for the structured-input experiment.</p><media xlink:href="jmir_v28i1e95162_app3.docx" xlink:title="DOCX File, 242 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Patient query&#x2013;style input variants and corresponding model outputs.</p><media xlink:href="jmir_v28i1e95162_app4.docx" xlink:title="DOCX File, 461 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Reviewer scoring dataset and scoring rubric.</p><media xlink:href="jmir_v28i1e95162_app5.xlsx" xlink:title="XLSX File, 45 KB"/></supplementary-material></app-group></back></article>