<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e92663</article-id><article-id pub-id-type="doi">10.2196/92663</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>A Supervised Fine-Tuned Large Language Model for Lifestyle Management in Patients With Prostate Cancer: Development and Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Jiang</surname><given-names>Fangyuan</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yang</surname><given-names>Qiuwen</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zheng</surname><given-names>Xin</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yin</surname><given-names>Nan</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Jiayi</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lan</surname><given-names>Tu</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wu</surname><given-names>Yuanjun</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lin</surname><given-names>Yuxin</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jiang</surname><given-names>Kui</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Chen</surname><given-names>Yalan</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Medical Informatics, School of Medicine, Nantong University</institution><addr-line>Qixiu Road 19#</addr-line><addr-line>Nantong</addr-line><addr-line>Jiangsu</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Urology, The First Affiliated Hospital of Soochow University</institution><addr-line>Suzhou</addr-line><addr-line>Jiangsu</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Balcarras</surname><given-names>Matthew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Li</surname><given-names>Huinian</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Chow</surname><given-names>James C L</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Zhang</surname><given-names>Xueli</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Yalan Chen, PhD, Department of Medical Informatics, School of Medicine, Nantong University, Qixiu Road 19#, Nantong, Jiangsu, 226001, China, +86 0513 8505 1891; <email>ylchen@ntu.edu.cn</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>7</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e92663</elocation-id><history><date date-type="received"><day>24</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>07</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>08</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Fangyuan Jiang, Qiuwen Yang, Xin Zheng, Nan Yin, Jiayi Zhang, Tu Lan, Yuanjun Wu, Yuxin Lin, Kui Jiang, Yalan Chen. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 21.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e92663"/><abstract><sec><title>Background</title><p>Lifestyle interventions for patients with prostate cancer have been shown to improve treatment adherence and quality of life. However, there remains a lack of large language models (LLMs) capable of delivering individualized and professional lifestyle recommendations under clearly defined medical safety boundaries and controlled evidence sources.</p></sec><sec><title>Objective</title><p>This study aimed to develop and evaluate a supervised fine-tuned LLM&#x2014;PCaPLMM_SFT (Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-Tuning)&#x2014;to support health literacy improvement and lifestyle self-management among patients with prostate cancer.</p></sec><sec sec-type="methods"><title>Methods</title><p>We searched English-language literature primarily from PubMed (February 2015 to February 2025) to build a structured lifestyle management knowledge base covering diet, physical activity, weight management, medication adherence, and psychological support. We used a retrieval-augmented generation pipeline to generate patient-style question-answer (QA) pairs from retrieved knowledge slices. Bilingual English-Chinese QA data were generated from English-language source evidence through patient-oriented reformulation and retrieval-augmented generation&#x2013;based answer generation, and independent English and Chinese test sets were constructed to assess bilingual QA performance. We trained Baichuan2-7B-Chat using a 2-stage strategy, consisting of continued pretraining, followed by supervised fine-tuning with low-rank adaptation. Model outputs were evaluated in 2 double-blind rounds by referee LLMs (Qwen3-Max and DeepSeek-R1) and compared with GPT-3.5-Turbo and the base Baichuan2-7B-Chat using 2500 queries across 5 lifestyle scenarios. Additionally, 3 domain experts conducted a blinded review of 50 QA samples (10 per scenario). We used the Mann-Whitney <italic>U</italic> test with effect size <italic>r</italic>, and Benjamini-Hochberg false discovery rate correction, and examined consistency using intraclass correlation coefficients.</p></sec><sec sec-type="results"><title>Results</title><p>Based on 2211 included publications, we constructed the PCaPLMM_SFT-Train dataset. The knowledge base yielded &#x003E;150,000 structured knowledge slices. After 2 rounds of review, we obtained 42,330 single-turn QA pairs and 3008 multiturn dialogues, and the supervised fine-tuning phase used 45,338 structured QA samples. In the dual-round referee LLM assessment, PCaPLMM_SFT consistently outperformed Baichuan2-7B-Chat across dimensions and showed comparable or superior performance to GPT-3.5-Turbo across 5 lifestyle scenarios. Consistency analyses indicated moderate to good agreement between referee models across rounds, supporting the robustness of the comparative evaluation.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>PCaPLMM_SFT demonstrates the feasibility of constructing a medical lifestyle&#x2013;focused LLM by integrating structured medical knowledge, QA-style training data, and a multilayer evaluation system. This framework provides a reproducible methodological foundation for evidence-based health education and lifestyle management and establishes groundwork for future evaluation in real-world health management settings.</p></sec></abstract><kwd-group><kwd>prostate cancer</kwd><kwd>lifestyle intervention</kwd><kwd>large language model</kwd><kwd>LLM-as-a-judge</kwd><kwd>retrieval-augmented generation</kwd><kwd>supervised fine-tuning</kwd><kwd>conversational AI</kwd><kwd>patient education</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Prostate cancer is the second most common malignancy among men worldwide, and its incidence and disease burden continue to rise. Evidence indicates that lifestyle factors can significantly influence the quality of life and disease trajectory for patients. These factors specifically include diet, weight management, and physical activity [<xref ref-type="bibr" rid="ref1">1</xref>]. Multiple randomized controlled trials and real-world intervention studies have validated the safety and efficacy of such lifestyle interventions on clinical outcomes, while also highlighting the persistent demand from both patients and health care professionals for individualized, evidence-based lifestyle guidance [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>With the rapid advancement of AI, large language models (LLMs) such as GPT-4 and BioRAG have shown strong potential in medical question answering, abstract generation, and clinical decision support [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Recent studies on LLM-based medical chatbots have further highlighted their potential in patient-facing health information delivery and communication, particularly in oncology-related patient education, while emphasizing risks related to hallucination, bias, privacy, and governance, as well as the need for curated knowledge sources, quality control, and continuous monitoring [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. For example, GPT-4 achieved above-average human-level performance in the United States Medical Licensing Examination, drawing wide attention to its potential clinical applications. However, these general-purpose models face two critical challenges in real-world medical scenarios: (1) the lack of structured domain-specific knowledge, which often leads to incomplete or incorrect retrieval of relevant evidence [<xref ref-type="bibr" rid="ref9">9</xref>]; and (2) factual inaccuracies or hallucinations, which raise concerns about their safe use in clinical communication and patient education [<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>Recent studies in critical care medicine have demonstrated that integrating domain knowledge bases (KBs) with AI can substantially improve prediction and decision support. For example, Yang et al [<xref ref-type="bibr" rid="ref11">11</xref>] developed a transformer-based time series framework for intensive care unit patients with sepsis, achieving high predictive accuracy and interpretability by using longitudinal physiological data. Similarly, Zhang et al [<xref ref-type="bibr" rid="ref12">12</xref>] proposed the MetaSepsisKnowHub platform, which combines retrieval-augmented generation (RAG) with curated biomarker knowledge to enhance LLM factual accuracy and clinical applicability. These efforts underscore the importance of domain-enhanced AI systems and provide valuable methodological inspiration. However, existing oncology-focused LLM systems and broader RAG-based medical LLM frameworks have mainly emphasized treatment-related information retrieval, radiotherapy education, molecular treatment recommendation, cancer progression prediction, or clinical decision support.</p><p>To our knowledge, no study has specifically developed and systematically evaluated an LLM fine-tuned for patient-facing lifestyle management in patients with prostate cancer. This task requires not only high comprehensibility and robust evidence alignment but also empathy and consistent safety safeguards in real-world dialogues. Authentic physician-patient conversation data are scarce. To address this challenge, it is essential to explore methods that generate high-quality, literature-driven training corpora and apply supervised fine-tuning (SFT) to enhance both the professionalism and the reliability of model outputs. Furthermore, Song et al [<xref ref-type="bibr" rid="ref13">13</xref>] have shown that medical LLMs often suffer from hallucinations and limited domain grounding, and that incorporating structured, literature-anchored knowledge&#x2014;particularly through knowledge graph&#x2013;based retrieval augmentation&#x2014;can markedly improve model accuracy, response consistency, and diagnostic reliability in complex clinical settings.</p><p>To address these unmet needs, we developed the PCaPLMM_SFT (Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-Tuning), a domain-specific LLM built on the Baichuan2-7B-Chat architecture [<xref ref-type="bibr" rid="ref14">14</xref>]. The model integrates literature-driven question-answer (QA) generation and supervised instruction tuning to build structured corpora across multiple lifestyle-related scenarios. Unlike prior oncology-focused or RAG-based medical LLM frameworks that mainly emphasize retrieval-grounded answering or clinical decision support, PCaPLMM_SFT focuses on patient-facing lifestyle management for patients with prostate cancer and uses RAG not only for evidence-grounded inference but also for transforming PubMed-based evidence into patient-style QA training data for domain-adaptive SFT.</p><p>Accordingly, this study aimed to construct and evaluate PCaPLMM_SFT as a proof-of-concept framework for AI-assisted lifestyle management in patients with prostate cancer. Specifically, we (1) built a structured, literature-driven QA corpus covering major lifestyle domains such as diet, exercise, weight management, medication adherence, and psychological support; (2) applied SFT to enhance the model&#x2019;s domain comprehension, empathy, and factual reliability; and (3) established a standardized, multidimensional evaluation system to assess evidence alignment, comprehensibility, relevance, empathy, and feasibility.</p><p>Together, these objectives provide a reproducible pathway for building disease-specific LLMs and assessing their potential in patient education and chronic disease management. The framework integrates literature-to-QA automation and structured knowledge organization, enabling scalable, evidence-based model development.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>We followed a multistage systematic workflow to develop the PCaPLMM_SFT (<xref ref-type="fig" rid="figure1">Figure 1</xref>). The process included 5 major steps: data preparation, construction of QA datasets with RAG and LLMs, model training, referee model design, and model evaluation. It forms a closed-loop pipeline from knowledge extraction to model optimization and assessment.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Development workflow of the PCaPLMM_SFT model. The workflow illustrates the end-to-end pipeline, encompassing literature retrieval, corpus cleaning and QA construction, model training, test-set validation, and iterative refinement based on evaluation feedback. BGE-M3: BAAI General Embedding M3; ICC: intraclass correlation coefficient; LLM: large language model; PCa: prostate cancer; PCaPLMM_SFT: Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-Tuning; QA: question-answer; RAG: retrieval-augmented generation; SFT: supervised fine-tuning.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92663_fig01.png"/></fig></sec><sec id="s2-2"><title>Data Preparation</title><p>The data were primarily obtained from English-language PubMed publications. Publications between February 2015 and February 2025 were retrieved using search terms related to prostate cancer and lifestyle management, including lifestyle, exercise, and diet. The full PubMed search strategy and eligibility criteria are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. In addition to original research papers, secondary research, such as systematic reviews and meta-analyses, was also included to supplement real-world evidence gaps.</p><p>To meet LLMs&#x2019; training requirements, all raw documents underwent standardized preprocessing before inclusion. Through multiple rounds of cleaning and semantic extraction, irrelevant content&#x2014;such as author information, correspondence addresses, DOIs, and copyright statements&#x2014;was removed. The remaining content was transformed into JSON format suitable for LLM learning (eg, &#x201C;text&#x201D;: &#x201C;paragraph content&#x201D;). For complex review papers and documents with irregular structures, we applied a hybrid automated manual annotation strategy to maintain semantic consistency and ensure high data quality, providing a reliable foundation for subsequent QA construction.</p></sec><sec id="s2-3"><title>Construction of the Prostate Cancer Lifestyle KB</title><p>Cleaned texts were segmented into minimal semantic units, or chunks, based on paragraph boundaries, topic continuity, and medical semantic integrity rather than arbitrary fixed-length splitting. For each chunk, source metadata, including publication information, lifestyle domain, original section, and evidence source, were retained to ensure traceability. Duplicated, noninformative, or structurally abnormal chunks were removed through automated filtering and manual spot-checking. Each chunk was embedded into high-dimensional semantic vectors using the BGE-M3-embedding model and stored in a FAISS (Facebook Artificial Intelligence Similarity Search)&#x2013;based vector database together with the original text and metadata. This process formed a structured KB enabling efficient and traceable semantic retrieval for downstream QA and inference tasks [<xref ref-type="bibr" rid="ref15">15</xref>].</p></sec><sec id="s2-4"><title>Retrieval-Augmented Generation</title><p>During QA generation, query vectors were generated from predefined keywords or semantic seeds. Top-k semantic retrieval (k=20) was performed on the KB to obtain relevant knowledge slices. The value of k=20 was selected as a pragmatic balance between evidence coverage and noise control: a smaller k may miss complementary evidence, whereas a larger k may introduce weakly relevant fragments and distract the model from the most relevant information [<xref ref-type="bibr" rid="ref16">16</xref>-<xref ref-type="bibr" rid="ref18">18</xref>]. Retrieved content and preconstructed questions were combined and fed into an LLM via prompting templates, producing answers grounded in medical evidence. RAG was used both for (1) retrieving evidence to generate QA training data and (2) retrieving evidence at inference time to ground responses. Retrieved slices were ranked by similarity, and no additional fixed similarity threshold was applied before response generation. The current system did not implement a formal evidence-grade weighting algorithm or automated conflict adjudication mechanism for inconsistent retrieved slices. When retrieved evidence was heterogeneous, the model was guided by evidence-oriented prompts to generate cautious and individualized responses while avoiding absolute conclusions. This approach was intended to reduce hallucinations and support the medical accuracy and contextual coherence of the outputs.</p></sec><sec id="s2-5"><title>Patient-Oriented Question Reformulation</title><p>Based on the prostate cancer lifestyle knowledge slices, core factual questions were extracted through medical templates. These base questions were then reformulated by Qwen3-Turbo into more natural, conversational, and patient-oriented queries. To prevent redundancy and semantic conflicts, semantic similarity filtering was applied to remove duplicate or stylistically inconsistent queries, thereby improving diversity and consistency.</p></sec><sec id="s2-6"><title>High-Quality Answer Generation</title><p>Each patient-style question and its corresponding knowledge slice were combined using a unified prompt template and input into Qwen3-Turbo for answer generation. Because Qwen3-Turbo was used as the primary QA generator, the generated corpus may reflect model-specific biases in linguistic style, knowledge coverage, safety-related phrasing, and reasoning patterns. To reduce unconstrained model generation, answers were generated from retrieved literature-based knowledge slices rather than solely from the model&#x2019;s parametric knowledge. The prompt required the model to prioritize evidence-based recommendations, precise terminology, and consistency with the retrieved medical evidence. The generated QA pairs were formatted in ShareGPT JSON style to ensure compatibility for SFT.</p><p>To ensure data quality, a manual and automated quality control process was applied. We manually reviewed generated QA pairs using a stratified sampling strategy; this manual review was designed as a stratified quality control audit rather than a complete manual validation of the entire corpus. The review was conducted independently by 2 researchers (XW and HH) with medical backgrounds. Manual assessment evaluated factual accuracy, evidence alignment, safety, linguistic clarity, structural validity, and actionability. This audit was used to identify common quality problems, refine editing criteria, and guide subsequent corpus cleaning and standardization. Interrater agreement metrics were calculated and are reported in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. Discrepant samples were further reviewed and used to guide correction, revision, or exclusion. This process was supplemented by automated checks, including semantic deduplication, terminology filtering, format validation, and logical conflict detection. Together, these procedures removed misleading, ungrammatical, or structurally abnormal samples, and improved the professionalism, naturalness, and standardization of the dataset [<xref ref-type="bibr" rid="ref19">19</xref>].</p></sec><sec id="s2-7"><title>Group-Based Semantic Partitioning to Prevent Train-Evaluation Contamination</title><p>To prevent train-test leakage, we adopted a base-question-group splitting strategy. Each base question and its 1&#x2010;2 paraphrased variants were grouped as a single unit and assigned entirely to either the training or evaluation set to avoid semantic overlap across datasets. In addition, multilevel deduplication and semantic similarity filtering were applied. All queries were encoded using BGE-M3, and cosine similarity between training and evaluation sets was computed. Cross-set query pairs with high similarity (near-paraphrase cases) were automatically removed, ensuring strict semantic isolation between training and evaluation data.</p></sec><sec id="s2-8"><title>Task Definition</title><p>To clarify the functional scope of the QA workflow, the QA task was defined as a composite generative process performed autonomously by PCaPLMM_SFT after continued pretraining and SFT. The task integrated two capabilities: (1) medical knowledge&#x2013;based reasoning and (2) generation of patient-oriented lifestyle recommendations. Inputs included a fixed system prompt (eg, &#x201C;You are PCaPLMM_SFT...provide professional and evidence-based responses&#x201D;), brief patient information, and a lifestyle-related query. The model first performed medical semantic interpretation and evidence-based reasoning using the structured literature KB and then converted the inferred content into clear, empathetic, and actionable recommendations. Outputs were restricted to a single evidence-grounded and medically accurate answer. Task performance was evaluated across the 5 dimensions: evidence alignment, comprehensibility, relevance, empathy, and feasibility.</p></sec><sec id="s2-9"><title>Training Strategy</title><p>This study adopted a 2-stage training strategy for PCaPLMM_SFT. Stage 1 performed parameter-efficient continued pretraining using low-rank adaptation adapters on the Baichuan2-7B-Chat backbone. Training was conducted in LLaMA-Factory with a 2048-token context window, the AdamW optimizer (learning rate 2&#x00D7;10<sup>&#x2013;</sup>&#x2074;), cosine scheduling with 200 warm-up steps, weight decay, gradient clipping (maximum norm=1.0), a per-device batch size of 4, and 8 gradient accumulation steps for 3 epochs. Stage 2 fine-tuned the stage 1 checkpoint using domain-adaptive supervised fine-tuning with low-rank adaptation under a 1024-token context window, AdamW (2&#x00D7;10<sup>&#x2013;</sup>&#x2075;), and a fixed 3-epoch schedule.</p></sec><sec id="s2-10"><title>Referee Model Construction</title><p>The referee models were implemented by accessing the application programming interfaces of DeepSeek-R1 and Qwen3-Max through an open model service platform. Using prompt engineering, we constructed structured, outline-based scoring instructions, forming an LLM-based referee mechanism. This LLM-as-a-judge framework was adopted based on prior studies showing that large LLMs can generate structured evaluation signals under standardized rubrics, although results may be affected by prompt sensitivity and model-specific preferences [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. The evaluation prompt templates incorporated knowledge slices, role definitions, scoring criteria, and explicit restrictions (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>) to support a consistent evaluation context and produce reproducible assessment outputs. To examine the robustness of the referee LLM outputs, we further assessed their interround stability, intermodel consistency, and human-LLM consistency.</p></sec><sec id="s2-11"><title>Multidimensional Scoring System</title><p>The referee mechanism assessed each candidate answer across 5 dimensions&#x2014;evidence alignment, comprehensibility, relevance, empathy, and feasibility&#x2014;using the detailed definitions and scoring criteria (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). A predefined safety rule was also applied to flag responses with potential patient safety risks. Because PCaPLMM_SFT was designed for patient education and lifestyle guidance rather than for diagnosis or treatment decision-making, unsafe outputs were operationalized using a threshold for potential patient harm: responses were considered unsafe if they could reasonably mislead patients toward harmful self-management behaviors or unsupervised changes in medical care. This predefined safety rule was used only for post hoc evaluation and not as a trigger for inserting disclaimers during generation. Unsafe labels were used for analysis rather than exclusion; flagged responses remained in the evaluation denominator for error analysis and reporting of representative failure cases. Responses were labeled &#x201C;unsafe&#x201D; when (1) the referee assigned a score of 1 on any subdimension of evidence alignment, indicating inconsistency with established findings; (2) the recommendation carried potential for direct patient harm, such as inappropriate medication suggestions, unsafe or contraindicated lifestyle or exercise advice, or omission of essential safety precautions; or (3) the answer encouraged patients to initiate, discontinue, or modify medical treatment without clinician supervision.</p></sec><sec id="s2-12"><title>Evaluation Procedure</title><p>A total of 2500 patient-style queries from English and Chinese test sets covering the 5 lifestyle management scenarios were used for evaluation. Candidate answers from all models were scored independently and blindly by the referee model in 2 rounds. Each dimension was rated on a 5-point Likert scale, ranging from &#x201C;strongly disagree&#x201D; (1) to &#x201C;strongly agree&#x201D; (5) (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>).</p></sec><sec id="s2-13"><title>Referee Consistency Analysis</title><p>To evaluate the stability of scores at the individual dimension level, each &#x201C;question&#x00D7; dimension&#x201D; pair was treated as an independent scoring unit. Intraclass correlation coefficient (ICC) (3,k) was computed using a 2-way mixed-effects model (consistency and average measures). Additionally, Pearson and Spearman correlations were calculated based on a concatenated score vector stacking the 5-dimension&#x2013;level scores for each question. This approach measures consistency on both continuous and ordinal scales.</p></sec><sec id="s2-14"><title>Human Expert Evaluation</title><p>To further validate the reliability of the referee LLM evaluations, we conducted a small-scale manual assessment by human experts. Three experts with &#x2265;10 years of experience in prostate cancer management (2 urologists and 1 prostate cancer research specialist) independently evaluated the model-generated responses. A total of 50 QA samples were randomly selected, covering 5 core lifestyle scenarios (10 samples per scenario), based on prior studies [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>]. This expert assessment served as a supplementary validation component to examine whether automated evaluation signals were broadly aligned with clinical judgment across key dimensions such as evidence alignment, safety, usefulness, feasibility, and patient-centered communication [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. All texts were fully deidentified, and experts received only the QA pairs without any model identifiers to minimize potential evaluation bias. The expert assessment criteria, scoring dimensions, and consistency evaluation methods were kept consistent with those used in the referee LLM evaluation. The sample expert evaluation form for assessing multiple model outputs on lifestyle management in patients with prostate cancer is provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p></sec><sec id="s2-15"><title>Statistical Analysis</title><p>All statistical analyses were performed using R (version 4.3.1; R Foundation for Statistical Computing) and Python (version 3.10; Python Software Foundation) with the SciPy (version 1.11; SciPy Community), pandas, and numpy libraries. Referee evaluation scores of candidate models were compared using the Mann-Whitney <italic>U</italic> test for 2 independent groups. In addition to the <italic>U</italic> statistic and 2-sided <italic>P</italic> values, standardized <italic>Z</italic> scores and effect sizes (<italic>r</italic>) were calculated, where <inline-formula><mml:math id="ieqn1"><mml:mi>r</mml:mi><mml:mtext>=</mml:mtext><mml:mrow><mml:mrow><mml:mtext>|</mml:mtext><mml:mi>Z</mml:mi><mml:mtext>|</mml:mtext></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:msqrt><mml:mi>N</mml:mi></mml:msqrt></mml:mrow></mml:mrow></mml:math></inline-formula> and <italic>N</italic> is the total sample size. Effect size interpretation followed Cohen thresholds: 0.1=small, 0.3=medium, and 0.5=large. The descriptive statistics (mean, SD) were summarized for each group, along with the corresponding 95% CIs. All multiple comparisons were corrected for the false discovery rate using the Benjamini-Hochberg procedure. A 2-sided <italic>P</italic>&#x003C;.05 was considered statistically significant.</p></sec><sec id="s2-16"><title>Ethical Considerations</title><p>This study did not involve patient recruitment, clinical intervention, or the collection or processing of patient-level identifiable information and therefore did not constitute a clinical trial. The study protocol, including the expert evaluation process, was reviewed and approved by the Medical Ethics Committee of Nantong University, with the approval number 2025&#x2010;25. The evaluation was conducted by invited domain experts, who were informed of the study purpose, evaluation procedures, and intended use of the results before participation. All experts participated voluntarily and provided informed consent. The evaluation results are reported only in aggregate form, and no personally identifiable information is presented in the manuscript.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>PCaPLMM_SFT-Train Dataset</title><p>Briefly, 7197 records were identified from PubMed, 7195 records were screened, 3188 reports were assessed for eligibility, and 2211 studies were finally included for KB construction. The PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flow diagram of our review is shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flowchart.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92663_fig02.png"/></fig><p>Based on 2211 included publications, we constructed the PCaPLMM_SFT-Train dataset tailored to prostate cancer lifestyle management scenarios (<xref ref-type="table" rid="table1">Table 1</xref>). The dataset comprised 1516 original studies and 695 secondary research papers. The content covered 5 core themes: diet and nutrition, physical activity, weight management, psychological support, and medication adherence. After automated cleaning and manual review, irrelevant content and potentially sensitive information were removed. The texts were then structured and labeled by theme to form the foundational corpus.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Composition of the PCaPLMM_SFT<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>-Train dataset.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dataset name</td><td align="left" valign="bottom">Type</td><td align="left" valign="bottom">Volume</td><td align="left" valign="bottom">Description</td></tr></thead><tbody><tr><td align="left" valign="top">Pretraining dataset</td><td align="left" valign="top">Text data</td><td align="left" valign="top">45,084 entries</td><td align="left" valign="top">Included 1516 original studies and 695 reviews for continued pretraining.</td></tr><tr><td align="left" valign="top">Single-turn dataset</td><td align="left" valign="top">Medical dialogues</td><td align="left" valign="top">42,330 pairs</td><td align="left" valign="top">Patient-style QA<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> pairs covering diet, exercise, weight management, psychological support, and medication adherence.</td></tr><tr><td align="left" valign="top">Multiturn dataset</td><td align="left" valign="top">Multiturn dialogues</td><td align="left" valign="top">3008 pairs</td><td align="left" valign="top">Extended multiturn interactions simulating lifestyle management scenarios.</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>PCaPLMM_SFT: Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-Tuning.</p></fn><fn id="table1fn2"><p><sup>b</sup>QA: question-answer.</p></fn></table-wrap-foot></table-wrap><p>Within the RAG framework, we used Qwen3-Turbo to automatically generate QA pairs. Questions were produced through template-based extraction, followed by paraphrasing to simulate patient-style queries. Answers were generated from knowledge slices retrieved from the structured KB and were manually screened to ensure accuracy and consistency. To further enhance bilingual understanding and response generation, we constructed a bilingual English-Chinese patient-style QA corpus from the same English-language source evidence through patient-oriented reformulation and RAG-based answer generation. Although non&#x2013;English literature databases were not independently searched, the generated corpus encompassed a broad range of lifestyle-related domains and incorporated culturally and regionally specific expressions. After 2 rounds of review, we obtained 42,330 single-turn QA pairs and 3008 multiturn dialogues.</p></sec><sec id="s3-2"><title>Automated QA Generation System</title><p>The QA generation system integrated structured medical literature knowledge with semantic retrieval and LLM-based generation to produce high-quality, automated patient-style QA content. The KB was constructed by importing unstructured medical texts, segmenting them into semantic vectors, and extracting metadata. A total of more than 150,000 knowledge slices were generated, covering all major lifestyle-related conversational scenarios. <xref ref-type="fig" rid="figure3">Figure 3</xref> illustrates the KB management interface, which includes literature import (<xref ref-type="fig" rid="figure3">Figure 3A</xref>), content editing (<xref ref-type="fig" rid="figure3">Figure 3B</xref>), structured indexing (<xref ref-type="fig" rid="figure3">Figure 3C</xref>), and semantic debugging modules (<xref ref-type="fig" rid="figure3">Figure 3D</xref>). Qwen3-Turbo was used to automatically generate patient-style questions, achieving interactive linkage between knowledge construction and data generation.</p><p>For each knowledge slice, the system generated 5 base questions aligned with lifestyle contexts. Each question was then paraphrased 1&#x2010;2 times by an LLM to enrich the linguistic diversity while preserving its core medical meaning. All questions were documented with their contents, scenario, generation type (original or paraphrased), and source reference. Examples are presented in <xref ref-type="table" rid="table2">Table 2</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Interface of the PCaPLMM_SFT knowledge base. This figure illustrates the core components of the knowledge base: (A) import and management, (B) semantic slicing, (C) structured indexing, and (D) similarity-based retrieval. PCa: prostate cancer; REDCap: Research Electronic Data Capture; SAS: Statistical Analysis System.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92663_fig03.png"/></fig><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Examples of patient-style question generation.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question</td><td align="left" valign="bottom">Scenario</td><td align="left" valign="bottom">Type</td><td align="left" valign="bottom">Source file</td></tr></thead><tbody><tr><td align="left" valign="top">What dietary changes should I make to potentially improve my prostate cancer outcome?</td><td align="left" valign="top">Diet</td><td align="left" valign="top">Original</td><td align="left" valign="top">1-s2.0-S0013935122007435-main.pdf</td></tr><tr><td align="left" valign="top">What modifications to my diet might enhance the prognosis of my prostate cancer?</td><td align="left" valign="top">Diet</td><td align="left" valign="top">Paraphrased</td><td align="left" valign="top">1-s2.0-S0013935122007435-main.pdf</td></tr><tr><td align="left" valign="top">Which dietary adjustments could potentially improve the results of my prostate cancer treatment?</td><td align="left" valign="top">Diet</td><td align="left" valign="top">Paraphrased</td><td align="left" valign="top">1-s2.0-S0013935122007435-main.pdf</td></tr><tr><td align="left" valign="top">Would regular aerobic exercise help manage symptoms and improve my quality of life after a prostate cancer diagnosis?</td><td align="left" valign="top">Exercise</td><td align="left" valign="top">Original</td><td align="left" valign="top">1-s2.0-S0013935122019193-main.pdf</td></tr><tr><td align="left" valign="top">Can consistent aerobic exercise alleviate symptoms and enhance my quality of life following a prostate cancer diagnosis?</td><td align="left" valign="top">Exercise</td><td align="left" valign="top">Paraphrased</td><td align="left" valign="top">1-s2.0-S0013935122019193-main.pdf</td></tr><tr><td align="left" valign="top">Is engaging in regular aerobic exercise beneficial for controlling symptoms and boosting my quality of life post&#x2013;prostate cancer diagnosis?</td><td align="left" valign="top">Exercise</td><td align="left" valign="top">Paraphrased</td><td align="left" valign="top">1-s2.0-S0013935122019193-main.pdf</td></tr><tr><td align="left" valign="top">Would increasing my physical activity improve my psychological well-being during prostate cancer treatment?</td><td align="left" valign="top">Psychological support</td><td align="left" valign="top">Original</td><td align="left" valign="top">12874_2021_Article_1486.pdf</td></tr><tr><td align="left" valign="top">Can engaging more in physical activities enhance my mental health while receiving prostate cancer treatment?</td><td align="left" valign="top">Psychological support</td><td align="left" valign="top">Paraphrased</td><td align="left" valign="top">12874_2021_Article_1486.pdf</td></tr><tr><td align="left" valign="top">Is boosting my level of exercise likely to positively impact my psychological state during prostate cancer therapy?</td><td align="left" valign="top">Psychological support</td><td align="left" valign="top">Paraphrased</td><td align="left" valign="top">12874_2021_Article_1486.pdf</td></tr></tbody></table></table-wrap><p>Following knowledge construction and question generation, we produced QA pairs that were explicitly grounded in medical evidence. For instance (<xref ref-type="fig" rid="figure4">Figure 4A</xref>), given the query &#x201C;Having recently learned that I have prostate cancer, I would like some guidance on diet that may help with lifestyle adjustments,&#x201D; the system retrieved 7 relevant research fragments (<xref ref-type="fig" rid="figure4">Figure 4B</xref>). The generated answer integrated dietary recommendations such as plant-based diets, reduced red meat and sugar intake, and appropriate fat consumption. The outputs were professional, structured, and standardized in ShareGPT JSON format (<xref ref-type="fig" rid="figure4">Figure 4C</xref>). To quantify dataset quality, we manually reviewed a stratified sample of 2400 QA pairs, representing 5.62% (2400/42,705) of the prefinal QA candidate corpus used for corpus auditing. Each sample was categorized by required editing level (<xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>): 50.71% (1217/2400) needed only minor phrasing or structural refinements, 24.58% (590/2400) required major revisions due to factual or evidence misalignment, 22.79% (547/2400) were accepted without changes, and 1.92% (46/2400) were discarded for severe or noncorrectable errors. The major revision rate reflects QA quality and not the residual error in the final SFT dataset. Samples requiring major revision were revised according to retrieved evidence, while noncorrectable or potentially misleading samples were removed. The validated editing criteria derived from this audit were then applied to clean the remaining dataset. Representative failure cases are provided in <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Example of knowledge-based question-answer (QA) generation. This figure shows the structured QA pair construction: (A) natural language input and retrieval, (B) extracted literature fragments, and (C) structured model output presented in the ShareGPT format. ADT: androgen deprivation therapy; PCa: prostate cancer; PCaPLMM_SFT: Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-Tuning.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92663_fig04.png"/></fig></sec><sec id="s3-3"><title>Model Training of PCaPLMM_SFT</title><p>The continued pretraining phase consisted of more than 2200 training steps. The loss decreased from approximately 2.9 to below 1.8 and stabilized after approximately 1600 steps (<xref ref-type="fig" rid="figure5">Figure 5A</xref>). A total of 23,263,936 tokens were used during this phase. In the subsequent SFT phase, 45,338 structured QA samples were used, with a total training volume of 60,788,864 tokens. The loss declined from approximately 2.0 to approximately 0.9 and converged after approximately 1750 steps (<xref ref-type="fig" rid="figure5">Figure 5B</xref>). The final model weights and configuration files have been publicly released on the Hugging Face platform (see the Data Availability section).</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Training loss curves of PCaPLMM_SFT (Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-Tuning). The curves display loss trajectories across (A) the continued pretraining phase and (B) the supervised fine-tuning phase.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92663_fig05.png"/></fig></sec><sec id="s3-4"><title>Referee Model Performance and Error Analysis</title><p>Across both evaluation rounds, PCaPLMM_SFT consistently outperformed Baichuan2-7B-Chat across all dimensions and surpassed GPT-3.5-Turbo in most metrics (<xref ref-type="table" rid="table3">Table 3</xref>). Under Qwen3-Max, PCaPLMM_SFT achieved evidence alignment mean scores of 4.156 (SD 0.591) and 4.155 (SD 0.591), significantly higher than Baichuan2-7B-Chat (|<italic>r</italic>|=0.601&#x2010;0.614; <italic>P</italic>&#x003C;.001) and GPT-3.5-Turbo (|<italic>r</italic>|=0.299&#x2010;0.305; <italic>P</italic>&#x003C;.001). The model also showed clear advantages in feasibility dimension (mean 3.191, SD 0.717; mean 3.202, SD 0.706), again outperforming both baselines (<italic>P</italic>&#x003C;.001).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Referee evaluation scores for candidate models<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimension (round)</td><td align="left" valign="bottom">PCaPLMM_SFT<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>, mean (SD), 95% CI</td><td align="left" valign="bottom">Baichuan2-7B-Chat, mean (SD), 95% CI</td><td align="left" valign="bottom">GPT-3.5-Turbo, mean (SD), 95% CI</td><td align="left" valign="bottom"><italic>P</italic><sub>1</sub> value (FDR<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup>)</td><td align="left" valign="bottom">|<italic>r</italic><sub>1</sub>|</td><td align="left" valign="bottom"><italic>P</italic><sub>2</sub> value (FDR)</td><td align="left" valign="bottom">|<italic>r</italic><sub>2</sub>|</td></tr></thead><tbody><tr><td align="left" valign="top">Qwen3-Max</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Evidence alignment &#x2460;</td><td align="left" valign="top">4.156 (0.591), 4.132-4.179</td><td align="left" valign="top">3.363 (0.906), 3.327-3.398</td><td align="left" valign="top">3.930 (0.522), 3.910-3.951</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.614</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.299</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">4.155 (0.591), 4.132-4.178</td><td align="left" valign="top">3.376 (0.902), 3.340-3.411</td><td align="left" valign="top">3.924 (0.518), 3.904-3.944</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.601</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.305</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Comprehensibility &#x2460;</td><td align="left" valign="top">4.561 (0.437), 4.544-4.578</td><td align="left" valign="top">3.857 (0.857), 3.824-3.891</td><td align="left" valign="top">4.460 (0.368), 4.445-4.474</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.607</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.185</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">4.571 (0.417), 4.555-4.588</td><td align="left" valign="top">3.855 (0.857), 3.821-3.889</td><td align="left" valign="top">4.456 (0.365), 4.442-4.471</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.607</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.221</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Relevance &#x2460;</td><td align="left" valign="top">2.899 (0.458), 2.881-2.917</td><td align="left" valign="top">1.959 (0.478), 1.940-1.978</td><td align="left" valign="top">2.832 (0.426), 2.815-2.849</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.821</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.109</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">2.911 (0.440), 2.894-2.928</td><td align="left" valign="top">1.962 (0.480), 1.943-1.981</td><td align="left" valign="top">2.829 (0.425), 2.812-2.846</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.823</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.148</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Empathy &#x2460;</td><td align="left" valign="top">4.563 (0.474), 4.544-4.581</td><td align="left" valign="top">4.116 (0.862), 4.082-4.150</td><td align="left" valign="top">4.580 (0.363), 4.565-4.594</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.424</td><td align="left" valign="top">.57</td><td align="left" valign="top">0.012</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">4.576 (0.447), 4.558-4.593</td><td align="left" valign="top">4.107 (0.864), 4.073-4.141</td><td align="left" valign="top">4.570 (0.366), 4.555-4.584</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.44</td><td align="left" valign="top">.14</td><td align="left" valign="top">0.031</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Feasibility &#x2460;</td><td align="left" valign="top">3.191 (0.717), 3.163-3.219</td><td align="left" valign="top">2.708 (0.707), 2.680-2.736</td><td align="left" valign="top">2.978 (0.544), 2.957-2.999</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.454</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.241</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">3.202 (0.706), 3.174-3.230</td><td align="left" valign="top">2.703 (0.711), 2.675-2.731</td><td align="left" valign="top">2.969 (0.552), 2.947-2.990</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.462</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.273</td></tr><tr><td align="left" valign="top">DeepSeek-R1</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Evidence alignment &#x2460;</td><td align="left" valign="top">4.345 (0.501), 4.325-4.365</td><td align="left" valign="top">3.740 (1.114), 3.696-3.783</td><td align="left" valign="top">3.915 (0.737), 3.886-3.944</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.491</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.446</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">4.310 (0.629), 4.285-4.335</td><td align="left" valign="top">3.544 (1.032), 3.504-3.584</td><td align="left" valign="top">3.885 (0.732), 3.857-3.914</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.663</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.428</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Comprehensibility &#x2460;</td><td align="left" valign="top">4.657 (0.468), 4.639-4.675</td><td align="left" valign="top">4.350 (0.851), 4.317-4.383</td><td align="left" valign="top">4.607 (0.456), 4.589-4.625</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.331</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.079</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">4.624 (0.570), 4.602-4.646</td><td align="left" valign="top">4.321 (0.884), 4.287-4.356</td><td align="left" valign="top">4.626 (0.437), 4.609-4.643</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.608</td><td align="left" valign="top">.80</td><td align="left" valign="top">0.005</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Relevance &#x2460;</td><td align="left" valign="top">3.739 (0.807), 3.707-3.770</td><td align="left" valign="top">3.000 (0.741), 2.971-3.029</td><td align="left" valign="top">3.260 (0.545), 3.239-3.282</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.578</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.422</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">3.242 (0.594), 3.219-3.266</td><td align="left" valign="top">2.993 (0.755), 2.963-3.023</td><td align="left" valign="top">3.259 (0.536), 3.238-3.280</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.851</td><td align="left" valign="top">.46</td><td align="left" valign="top">0.019</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Empathy &#x2460;</td><td align="left" valign="top">4.634 (0.491), 4.615-4.654</td><td align="left" valign="top">4.463 (0.686), 4.436-4.490</td><td align="left" valign="top">4.698 (0.401), 4.683-4.714</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.219</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.095</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">4.637 (0.547), 4.615-4.658</td><td align="left" valign="top">4.431 (0.736), 4.402-4.460</td><td align="left" valign="top">4.703 (0.383), 4.688-4.718</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.462</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.101</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Feasibility &#x2460;</td><td align="left" valign="top">3.968 (0.706), 3.940-3.995</td><td align="left" valign="top">3.392 (0.846), 3.358-3.425</td><td align="left" valign="top">3.373 (0.621), 3.348-3.397</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.479</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.53</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">3.741 (0.649), 3.715-3.766</td><td align="left" valign="top">3.283 (0.806), 3.251-3.314</td><td align="left" valign="top">3.298 (0.584), 3.275-3.321</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.737</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.483</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup><italic>P</italic><sub>1</sub> and |<italic>r</italic><sub>1</sub>| represent the comparative results between PCaPLMM_SFT and Baichuan2-7B-Chat, whereas <italic>P</italic><sub>2</sub> and |<italic>r</italic><sub>2</sub>| correspond to the comparison between PCaPLMM_SFT and GPT-3.5-Turbo. <italic>P</italic> values are derived from 2-sided Mann-Whitney <italic>U</italic> tests and adjusted using the Benjamini-Hochberg procedure to control the false discovery rate. |<italic>r</italic>| denotes the absolute effect size, calculated from the standardized <italic>Z</italic> statistic and accompanied by 95% CIs estimated via Fisher <italic>Z</italic> transformation.</p></fn><fn id="table3fn2"><p><sup>b</sup>PCaPLMM_SFT: Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-Tuning.</p></fn><fn id="table3fn3"><p><sup>c</sup>FDR: false discovery rate.</p></fn></table-wrap-foot></table-wrap><p>DeepSeek-R1 produced consistent findings: PCaPLMM_SFT showed significant gains in evidence alignment (mean 4.345, SD 0.501; mean 4.310, SD 0.629; <italic>P</italic>&#x003C;.001) and delivered top performance in comprehensibility (4.624&#x2010;4.657) and empathy (4.634&#x2010;4.637), with significant margins over Baichuan2-7B-Chat (|<italic>r</italic>|&#x2265;0.219; <italic>P</italic>&#x003C;.001).</p><p>Across the 2 referee LLMs, interrater and interround consistency analyses showed overall moderate to good agreement (<xref ref-type="table" rid="table4">Table 4</xref>). In round 1, Qwen3-Max and DeepSeek-R1 demonstrated moderate interrater consistency (ICC(3,k)=0.540; Pearson <italic>r</italic>=0.382; Spearman <italic>&#x03C1;</italic>=0.415), indicating generally comparable scoring patterns despite differences in scoring sensitivity. In round 2, interrater agreement was higher than in the first round (ICC(3,k)=0.747; Pearson <italic>r</italic>=0.600; Spearman <italic>&#x03C1;</italic>=0.655), reflecting a consistent scoring tendency between the 2 models when independently evaluating the same QA scenarios.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Interrater and interround consistency of large language model referee evaluations<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparison and metric type</td><td align="left" valign="bottom">Estimate</td><td align="left" valign="bottom">95% CI</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Qwen versus DeepSeek (round 1)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ICC<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup>(3,k)</td><td align="left" valign="top">0.540</td><td align="left" valign="top">0.530-0.550</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pearson <italic>r</italic></td><td align="left" valign="top">0.382</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Spearman <italic>&#x03C1;</italic></td><td align="left" valign="top">0.415</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Qwen versus DeepSeek (round 2)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ICC(3,k)</td><td align="left" valign="top">0.747</td><td align="left" valign="top">0.740-0.750</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pearson <italic>r</italic></td><td align="left" valign="top">0.600</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Spearman <italic>&#x03C1;</italic></td><td align="left" valign="top">0.655</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Round 1 versus round 2 (Qwen only)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ICC(3,k)</td><td align="left" valign="top">0.816</td><td align="left" valign="top">0.810-0.820</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pearson <italic>r</italic></td><td align="left" valign="top">0.690</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Spearman <italic>&#x03C1;</italic></td><td align="left" valign="top">0.719</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Round 1 versus round 2 (DeepSeek only)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ICC(3,k)</td><td align="left" valign="top">0.507</td><td align="left" valign="top">0.490-0.520</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pearson <italic>r</italic></td><td align="left" valign="top">0.343</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Spearman <italic>&#x03C1;</italic></td><td align="left" valign="top">0.378</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>ICC(3,k) was calculated using a 2-way mixed-effects model (consistency type and average measures) under the assumption of fixed raters. Pearson <italic>r</italic> and Spearman <italic>&#x03C1;</italic> were computed using a concatenated scoring vector formed by stacking the 5 evaluation dimensions for each question, allowing assessment of cross-dimensional agreement.</p></fn><fn id="table4fn2"><p><sup>b</sup>ICC: intraclass correlation coefficient.</p></fn><fn id="table4fn3"><p><sup>c</sup>Not available.</p></fn></table-wrap-foot></table-wrap><p>Within-model interround consistency showed clear differences: Qwen3-Max exhibited high reproducibility (ICC(3,k)=0.816; Pearson <italic>r</italic>=0.690; Spearman <italic>&#x03C1;</italic>=0.719), whereas DeepSeek-R1 demonstrated only moderate stability (ICC(3,k)=0.507; Pearson <italic>r</italic>=0.343; Spearman <italic>&#x03C1;</italic>=0.378). Overall, although the 2 referee models differed in scoring variability, their scoring trends remained reasonably aligned, supporting the robustness of the comparative evaluation across repeated assessments.</p><p>Across 2 independent generations, low-scoring outputs were grouped into 4 error categories (<xref ref-type="table" rid="table5">Table 5</xref>). PCaPLMM_SFT showed the lowest error rates overall, with evidence misalignment: 3.68%, overly general or insufficiently actionable advice: 9.28%, imprecise or insufficiently patient-centered language: 5.68%, and hallucinations: 5.32%. Baichuan2-7B-Chat demonstrated substantially higher error rates in all categories, particularly evidence misalignment (20.66%) and hallucinations (12.88%). GPT-3.5-Turbo showed intermediate performance, with a notably low rate of patient-centeredness errors (1.42%) (<xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>). Evidence misalignment and hallucinations were treated as potentially risky error categories because they may mislead patients or provide unsupported lifestyle recommendations, although actual patient harm could not be directly assessed in this offline evaluation without real patient interactions. Overall, PCaPLMM_SFT presented fewer potentially risky errors, suggesting better evidence alignment and reliability in generating lifestyle recommendations.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Representative failure cases of PCaPLMM_SFT<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Error type</td><td align="left" valign="bottom">Example question</td><td align="left" valign="bottom">Original answer</td><td align="left" valign="bottom">Problem identified</td></tr></thead><tbody><tr><td align="left" valign="top">Evidence misalignment</td><td align="left" valign="top">How much physical activity should I engage in weekly to support my prostate cancer treatment?</td><td align="left" valign="top">&#x201C;You should perform at least 150 minutes of moderate-intensity or 75 minutes of vigorous-intensity aerobic activity per week, plus 2&#x2010;3 sessions of resistance training...androgen deprivation therapy may increase fat mass and decrease muscle mass, so exercise is helpful...&#x201D;</td><td align="left" valign="top">The model incorrectly applied general population World Health Organization aerobic thresholds to a prostate cancer context. Evidence shows that men undergoing treatment&#x2014;especially those on androgen deprivation therapy&#x2014;benefit more from resistance-focused programs with individualized aerobic activity. Thus, the original answer overstated the suitability of population-level recommendations and overlooked cancer-specific safety and personalized exercise needs.</td></tr><tr><td align="left" valign="top">Insufficiently actionable advice</td><td align="left" valign="top">&#x201C;What should I eat after prostate cancer surgery?&#x201D;</td><td align="left" valign="top">&#x201C;Maintain a balanced diet, eat vegetables and fruits, avoid fatty foods.&#x201D;</td><td align="left" valign="top">Overall, lacks actionable details such as frequency, quantities, and specific limits.</td></tr><tr><td align="left" valign="top">Imprecise or non&#x2013;patient-centered expression</td><td align="left" valign="top">&#x201C;I feel anxious at night after surgery. What should I do?&#x201D;</td><td align="left" valign="top">&#x201C;Postoperative anxiety is a common psychiatric reaction. You need to improve your mood response.&#x201D;</td><td align="left" valign="top">Language is technical and non&#x2013;patient-centered; lacks reassurance and evidence-based coping strategies.</td></tr><tr><td align="left" valign="top">Hallucinations</td><td align="left" valign="top">How does maintaining a balanced diet rich in fruits and vegetables affect prostate cancer progression?</td><td align="left" valign="top">&#x201C;A systematic review and meta-analysis. Prostate Cancer and Prostatic Diseases, 2016. 17(4): 357&#x2010;68.&#x201D;</td><td align="left" valign="top">The model did not answer the question and instead produced a fabricated or irrelevant citation. This represents a hallucination where the model outputs a pseudoreference unrelated to the user query and fails to provide evidence-based dietary guidance. Such hallucinations may mislead readers by mimicking academic citations without substantive content.</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>PCaPLMM_SFT: Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-Tuning.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Human Expert Ratings and Consistency With Referee LLMs</title><p>A total of 3 experts independently evaluated 50 deidentified QA samples, and their ratings were compared with those of the 2 LLM referee models (<xref ref-type="table" rid="table6">Table 6</xref>). Interexpert agreement was moderate, with ICC(3,k) ranging from 0.332 to 0.431 across expert pairs, and corresponding Pearson correlations of 0.394&#x2010;0.471 and Spearman correlations of 0.380&#x2010;0.511.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Interrater reliability between human experts and LLM<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup> referee models.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparison and metric type</td><td align="left" valign="bottom">Estimate</td><td align="left" valign="bottom">95% CI</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Expert 1 and Expert 2</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ICC<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup>(3,k)</td><td align="left" valign="top">0.413</td><td align="left" valign="top">0.360-0.460</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pearson <italic>r</italic></td><td align="left" valign="top">0.471</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Spearman <italic>&#x03C1;</italic></td><td align="left" valign="top">0.511</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Expert 1 and Expert 3</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ICC(3,k)</td><td align="left" valign="top">0.332</td><td align="left" valign="top">0.270-0.390</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pearson <italic>r</italic></td><td align="left" valign="top">0.394</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Spearman <italic>&#x03C1;</italic></td><td align="left" valign="top">0.380</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Expert 2 and Expert 3</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ICC(3,k)</td><td align="left" valign="top">0.431</td><td align="left" valign="top">0.380-0.480</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pearson <italic>r</italic></td><td align="left" valign="top">0.458</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Spearman <italic>&#x03C1;</italic></td><td align="left" valign="top">0.427</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Qwen versus DeepSeek</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ICC(3,k)</td><td align="left" valign="top">0.689</td><td align="left" valign="top">0.660-0.710</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pearson <italic>r</italic></td><td align="left" valign="top">0.527</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Spearman <italic>&#x03C1;</italic></td><td align="left" valign="top">0.506</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Human Experts versus LLM Referees</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ICC(3,k)</td><td align="left" valign="top">0.474</td><td align="left" valign="top">0.410-0.530</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pearson <italic>r</italic></td><td align="left" valign="top">0.367</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Spearman <italic>&#x03C1;</italic></td><td align="left" valign="top">0.365</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table6fn2"><p><sup>b</sup>ICC: intraclass correlation coefficient.</p></fn><fn id="table6fn3"><p><sup>c</sup>Not available.</p></fn></table-wrap-foot></table-wrap><p>The 2 referee LLMs (Qwen3-Max vs DeepSeek-R1) demonstrated higher intermodel agreement (ICC(3,k)=0.689; Pearson <italic>r</italic>=0.527; Spearman <italic>&#x03C1;</italic>=0.506), indicating a relatively consistent scoring pattern between models. When comparing human experts with LLM referees, consistency was moderate (ICC(3,k)=0.474, 95% CI 0.410&#x2010;0.530), with Pearson <italic>r</italic>=0.367 and Spearman <italic>&#x03C1;</italic>=0.365. These results suggest that although humans and LLM referees share similar overall ranking tendencies across QA outputs, divergence remains in fine-grained scoring decisions. Human experts&#x2019; evaluation scores for the candidate models are provided in <xref ref-type="supplementary-material" rid="app9">Multimedia Appendix 9</xref>.</p></sec><sec id="s3-6"><title>Use Case Scenarios</title><p>We selected 5 common lifestyle management themes&#x2014;diet, exercise, weight management, medication adherence, and psychological support&#x2014;to construct the representative patient-style QA scenarios. These were used to evaluate the adaptability and conversational ability of PCaPLMM_SFT in natural patient queries (<xref ref-type="table" rid="table7">Table 7</xref>). <xref ref-type="table" rid="table7">Table 7</xref> illustrates the model responses across these scenarios and not the best-case outputs. The full evaluation set of 2500 patient-style queries and representative failure cases and error-category analyses are provided in <xref ref-type="table" rid="table5">Table 5</xref> and <xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>.</p><table-wrap id="t7" position="float"><label>Table 7.</label><caption><p>Example outputs from PCaPLMM_SFT<sup><xref ref-type="table-fn" rid="table7fn1">a</xref></sup> across 5 lifestyle scenarios<sup><xref ref-type="table-fn" rid="table7fn2">b</xref></sup>.</p></caption><table id="table7" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Scenario</td><td align="left" valign="bottom">Patient query</td><td align="left" valign="bottom">Model response (excerpt)</td></tr></thead><tbody><tr><td align="left" valign="top">Diet</td><td align="left" valign="top">How can I adjust my diet to better manage weight and maintain muscle strength while undergoing androgen deprivation therapy (ADT)?</td><td align="left" valign="top">During ADT, eat whole grains, fruits, vegetables, lean proteins (chicken, fish, beans, legumes), and healthy fats (avocados, nuts, seeds, olive oil); increase calcium and vitamin D from dairy, leafy greens, fish or supplements; drink plenty of water; limit processed foods, sugary drinks, and high-fat snacks; control portions with smaller plates; and combine diet with resistance training. Always consult with your healthcare provider before making significant changes to your diet or exercise regimen.</td></tr><tr><td align="left" valign="top">Exercise</td><td align="left" valign="top">What types of exercise are recommended for prostate cancer patients at different stages?</td><td align="left" valign="top">Recommended exercises include aerobic training &#x2265;3 times/week, resistance training twice weekly, balance/core exercises, and impact-loading workouts such as jogging. Intensity and frequency should be tailored to treatment phase and fitness.</td></tr><tr><td align="left" valign="top">Psychological support</td><td align="left" valign="top">I&#x2019;ve been feeling very anxious and stressed since being diagnosed with prostate cancer. What kind of psychological support can help me cope with this?</td><td align="left" valign="top">Support groups, individual counseling, Cognitive Behavioral Therapy, mindfulness, and online resources may help. Each patient is unique, so finding the right combination of emotional, social, and psychological support is essential for mental well-being.</td></tr><tr><td align="left" valign="top">Medication adherence</td><td align="left" valign="top">If I miss a dose or don&#x2019;t take my prostate cancer meds on time, is that a big deal? What should I be worried about?</td><td align="left" valign="top">Non-adherence can reduce treatment efficacy, increase recurrence risk, and worsen outcomes. If facing challenges with your medication, talk to your provider to find solutions, such as adjusting schedules or using reminders.</td></tr><tr><td align="left" valign="top">Weight management</td><td align="left" valign="top">As a prostate cancer patient, I&#x2019;m concerned that my current weight might increase my risk of heart problems. What are some safe ways for me to manage my BMI?</td><td align="left" valign="top">Combine healthy eating, exercise, sleep hygiene, and stress management. Consult healthcare professionals to develop a personalized plan for safe weight management and cardiovascular protection.</td></tr></tbody></table><table-wrap-foot><fn id="table7fn1"><p><sup>a</sup>PCaPLMM_SFT: Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-Tuning.</p></fn><fn id="table7fn2"><p><sup>b</sup>Responses shown in <xref ref-type="table" rid="table7">Table 7</xref> are excerpts from automatically generated outputs. Full responses of the example outputs from PCaPLMM_SFT across 5 lifestyle scenarios are available in <xref ref-type="supplementary-material" rid="app10">Multimedia Appendix 10</xref>. All answers underwent human review for semantic coherence, medical accuracy, and inclusion of safety disclaimers.</p></fn></table-wrap-foot></table-wrap><p>In the dietary management scenario, the query &#x201C;How can I adjust my diet to better manage weight and maintain muscle strength while undergoing androgen deprivation therapy (ADT)?&#x201D; elicited model-generated outputs recommending the consumption of whole grains, fruits, vegetables, lean proteins, and healthy fats, together with calcium and vitamin D supplementation. The model further advised limiting processed foods, sugary beverages, and high-fat snacks, combining dietary changes with resistance training, and underscored the need for individualized planning under physician supervision.</p><p>In exercise-related tasks, the model generated recommendations tailored to treatment phase and physical fitness level. For psychological support, the generated responses conveyed empathy and humanistic care. For medication adherence and weight management, the answers incorporated behavioral interventions, reminders, and risk prevention strategies. Overall, the model frequently generated safety-related statements such as &#x201C;consult your physician&#x201D; and &#x201C;tailor interventions to individual assessment.&#x201D; These statements reflected medical communication patterns learned during SFT and cautious response generation guided by the system prompt, rather than deterministic rule-based safety guardrails. PCaPLMM_SFT consistently outperformed Baichuan2-7B-Chat and demonstrated comparable or superior performance to GPT-3.5-Turbo across all 5 lifestyle scenarios in 2 independent evaluation rounds conducted by Qwen3-Max and DeepSeek-R1 (<xref ref-type="table" rid="table8">Table 8</xref>).</p><table-wrap id="t8" position="float"><label>Table 8.</label><caption><p>Performance evaluation of 3 models across 5 lifestyle scenarios by referee LLMs<sup><xref ref-type="table-fn" rid="table8fn1">a</xref></sup>.</p></caption><table id="table8" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Scenario (round)</td><td align="left" valign="bottom">PCaPLMM_SFT<sup><xref ref-type="table-fn" rid="table8fn2">b</xref></sup>, mean (SD), 95% CI</td><td align="left" valign="bottom">Baichuan2-7B-Chat, mean (SD), 95% CI</td><td align="left" valign="bottom">GPT-3.5-Turbo, mean (SD), 95% CI</td><td align="left" valign="bottom"><italic>P</italic><sub>1</sub> value (FDR)<sup><xref ref-type="table-fn" rid="table8fn3">c</xref></sup></td><td align="left" valign="bottom">|r<sub>1</sub>|</td><td align="left" valign="bottom"><italic>P</italic><sub>2</sub> value (FDR)</td><td align="left" valign="bottom">|<italic>r</italic><sub>2</sub>|</td></tr></thead><tbody><tr><td align="left" valign="top">Qwen3-Max</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Diet &#x2460;</td><td align="left" valign="top">18.979 (2.209), 18.785-19.173</td><td align="left" valign="top">16.743 (3.833), 16.407-17.080</td><td align="left" valign="top">18.916 (1.528), 18.782-19.050</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.484</td><td align="left" valign="top">.27</td><td align="left" valign="top">0.053</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">19.007 (2.186), 18.815-19.199</td><td align="left" valign="top">16.822 (3.548), 16.510-17.134</td><td align="left" valign="top">18.856 (1.496), 18.724-18.987</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.479</td><td align="left" valign="top">.86</td><td align="left" valign="top">0.008</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Exercise &#x2460;</td><td align="left" valign="top">20.063 (2.321), 19.859-20.267</td><td align="left" valign="top">15.968 (3.650), 15.647-16.288</td><td align="left" valign="top">18.653 (1.552), 18.517-18.790</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.694</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.485</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">20.098 (2.295), 19.896-20.299</td><td align="left" valign="top">16.013 (3.618), 15.695-16.331</td><td align="left" valign="top">18.568 (1.583), 18.429-18.707</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.704</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.578</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>PS<sup><xref ref-type="table-fn" rid="table8fn4">d</xref></sup> &#x2460;</td><td align="left" valign="top">19.147 (2.457), 18.931-19.363</td><td align="left" valign="top">15.445 (3.853), 15.106-15.783</td><td align="left" valign="top">18.813 (1.697), 18.664-18.962</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.627</td><td align="left" valign="top">.001</td><td align="left" valign="top">0.16</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">19.222 (2.396), 19.011-19.432</td><td align="left" valign="top">15.378 (4.067), 15.020-15.735</td><td align="left" valign="top">18.789 (1.733), 18.637-18.942</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.64</td><td align="left" valign="top">.02</td><td align="left" valign="top">0.104</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MA<sup><xref ref-type="table-fn" rid="table8fn5">e</xref></sup> &#x2460;</td><td align="left" valign="top">19.280 (1.720), 19.129-19.431</td><td align="left" valign="top">15.759 (2.647), 15.527-15.992</td><td align="left" valign="top">19.072 (1.287), 18.959-19.185</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.746</td><td align="left" valign="top">.001</td><td align="left" valign="top">0.157</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">19.246 (1.669), 19.100-19.393</td><td align="left" valign="top">15.857 (2.481), 15.639-16.075</td><td align="left" valign="top">18.959 (1.278), 18.846-19.071</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.745</td><td align="left" valign="top">.008</td><td align="left" valign="top">0.125</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>WM<sup><xref ref-type="table-fn" rid="table8fn6">f</xref></sup> &#x2460;</td><td align="left" valign="top">19.379 (2.319), 19.175-19.583</td><td align="left" valign="top">16.099 (3.144), 15.822-16.375</td><td align="left" valign="top">18.443 (1.740), 18.290-18.596</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.659</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.338</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">19.503 (2.066), 19.321-19.684</td><td align="left" valign="top">15.944 (3.462), 15.640-16.248</td><td align="left" valign="top">18.566 (1.761), 18.412-18.721</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.641</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.312</td></tr><tr><td align="left" valign="top">DeepSeek-R1</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Diet &#x2460;</td><td align="left" valign="top">21.368 (1.836), 21.207-21.529</td><td align="left" valign="top">18.979 (3.901), 18.636-19.321</td><td align="left" valign="top">20.079 (1.674), 19.932-20.226</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.528</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.451</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">20.438 (2.258), 20.239-20.636</td><td align="left" valign="top">18.665 (3.766), 18.335-18.995</td><td align="left" valign="top">20.023 (1.597), 19.883-20.163</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.645</td><td align="left" valign="top">.02</td><td align="left" valign="top">0.116</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Exercise &#x2460;</td><td align="left" valign="top">21.359 (2.376), 21.151-21.568</td><td align="left" valign="top">18.497 (4.511), 18.101-18.893</td><td align="left" valign="top">19.565 (2.529), 19.343-19.788</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.528</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.462</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">20.327 (2.776), 20.083-20.571</td><td align="left" valign="top">18.219 (4.166), 17.854-18.584</td><td align="left" valign="top">19.479 (2.488), 19.261-19.698</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.701</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.245</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>PS &#x2460;</td><td align="left" valign="top">21.515 (1.875), 21.351-21.680</td><td align="left" valign="top">19.483 (3.058), 19.214-19.751</td><td align="left" valign="top">19.990 (2.248), 19.793-20.188</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.515</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.45</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">20.942 (2.345), 20.736-21.148</td><td align="left" valign="top">18.888 (3.362), 18.593-19.183</td><td align="left" valign="top">19.905 (2.041), 19.725-20.084</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.76</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.339</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MA &#x2460;</td><td align="left" valign="top">21.276 (2.137), 21.089-21.464</td><td align="left" valign="top">19.112 (3.211), 18.830-19.394</td><td align="left" valign="top">20.188 (2.099), 20.003-20.372</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.527</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.35</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">20.521 (1.665), 20.374-20.667</td><td align="left" valign="top">18.770 (3.676), 18.447-19.092</td><td align="left" valign="top">20.082 (2.112), 19.896-20.267</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.841</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.187</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>WM &#x2460;</td><td align="left" valign="top">21.196 (2.453), 20.980-21.411</td><td align="left" valign="top">18.654 (3.874), 18.313-18.994</td><td align="left" valign="top">19.445 (2.719), 19.206-19.684</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.495</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.431</td></tr><tr><td align="left" valign="top">&#x2461;</td><td align="left" valign="top">20.540 (2.427), 20.327-20.753</td><td align="left" valign="top">18.319 (3.795), 17.986-18.651</td><td align="left" valign="top">19.369 (2.481), 19.151-19.587</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.736</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.338</td></tr></tbody></table><table-wrap-foot><fn id="table8fn1"><p><sup>a</sup><italic>P</italic><sub>1</sub> and |<italic>r</italic><sub>1</sub>| represent the comparative results between PCaPLMM_SFT and Baichuan2-7B-Chat, whereas <italic>P</italic><sub>2</sub> and |<italic>r</italic><sub>2</sub>| correspond to the comparison between PCaPLMM_SFT and GPT-3.5-Turbo. <italic>P</italic> values are derived from 2-sided Mann&#x2013;Whitney <italic>U</italic> tests and adjusted using the Benjamini&#x2013;Hochberg procedure to control the false discovery rate. |<italic>r</italic>| denotes the absolute effect size, calculated from the standardized <italic>Z</italic> statistic and accompanied by 95% confidence intervals estimated via Fisher <italic>Z</italic> transformation.</p></fn><fn id="table8fn2"><p><sup>b</sup>PCaPLMM_SFT: Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-tuning.</p></fn><fn id="table8fn3"><p><sup>c</sup>FDR: false discovery rate.</p></fn><fn id="table8fn4"><p><sup>d</sup>PS: psychological support.</p></fn><fn id="table8fn5"><p><sup>e</sup>MA: medication adherence.</p></fn><fn id="table8fn6"><p><sup>f</sup>WM: weight management.</p></fn></table-wrap-foot></table-wrap><p>Under Qwen3-Max evaluation, PCaPLMM_SFT achieved significantly higher scores than Baichuan2-7B-Chat across all scenarios (diet, exercise, psychological support, medication adherence, and weight management; <italic>P</italic>&#x003C;.001), with medium-to-large effect sizes (|<italic>r</italic>| ranging from 0.479 to 0.746). Compared with GPT-3.5-Turbo, PCaPLMM_SFT showed clear advantages in specific scenarios, particularly exercise, psychological support, and weight management (|<italic>r</italic>| ranging from 0.104 to 0.578), while maintaining comparable performance in others (eg, diet).</p><p>Under DeepSeek-R1 evaluation, the superiority of PCaPLMM_SFT over Baichuan2-7B-Chat was consistent and significant across all 5 scenarios, with the largest effect sizes observed in psychological support and medication adherence (<italic>P</italic>&#x003C;.001; |<italic>r</italic>| ranging from 0.515 to 0.841). Comparisons against GPT-3.5-Turbo showed a similar favorable pattern, with PCaPLMM_SFT significantly outperforming the benchmark in most tasks (<italic>P</italic>&#x003C;.001; |<italic>r</italic>| ranging from 0.116 to 0.462).</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this study, we developed and implemented PCaPLMM_SFT, a SFT LLM designed to support healthy lifestyle management for patients with prostate cancer. By integrating a structured literature-based KB, semantic retrieval, and instruction tuning, we generated high-quality QA pairs covering multiple lifestyle-related dialogue scenarios. Furthermore, 2 referee models were introduced for multidimensional evaluation. Results indicated that our model significantly outperformed the general-purpose models in key dimensions such as evidence alignment and comprehensibility, highlighting its potential for patient-clinician communication and personalized patient education. In addition to the automated referee evaluation, a small-scale human expert assessment was conducted to further validate the reliability of the referee model framework. Across both referee LLMs and 3 human experts, PCaPLMM_SFT was consistently ranked as the top-performing model in all lifestyle domains. Although the level of agreement varied across evaluators, the overall ranking trend remained stable, supporting the robustness of the comparative results.</p><p>For the base architecture, we selected Baichuan2-7B-Chat, which offers a balance between parameter scale and open accessibility, enabling targeted optimization and potential local deployment [<xref ref-type="bibr" rid="ref14">14</xref>]. Unlike prior models that relied on authentic physician-patient dialogues for fine-tuning (eg, HuatuoGPT [<xref ref-type="bibr" rid="ref26">26</xref>]) or expert-curated annotations (eg, Med-PaLM [<xref ref-type="bibr" rid="ref23">23</xref>]), this study proposes an automated, literature-driven framework for training data generation. Through semantic chunking, structured indexing, and prompt-based reformulation, we constructed a large-scale, low-cost, and high-quality training corpus covering 5 major lifestyle scenarios. This approach not only enhances the controllability of medical knowledge but also addresses the long-standing challenges of data scarcity due to privacy restrictions and limited willingness for data sharing.</p><p>At the task level, PCaPLMM_SFT demonstrates an improved capacity for understanding and generating patient-style language. Compared with the document-focused generative models such as RadOnc-GPT [<xref ref-type="bibr" rid="ref27">27</xref>], PCaPLMM_SFT emphasizes flexibility and empathy in open-ended dialogue, enabling multiturn interactions that incorporated emotional recognition, health guidance, and behavioral recommendations. These features improve communication effectiveness and user engagement in chronic disease management.</p><p>Traditional evaluation of medical QA systems has relied heavily on expert ratings, which are costly, subjective, and difficult to scale. With the emergence of the &#x201C;LLM-as-a-Judge&#x201D; paradigm [<xref ref-type="bibr" rid="ref28">28</xref>-<xref ref-type="bibr" rid="ref30">30</xref>], recent studies have demonstrated that LLM-based referees can provide stable, multidimensional evaluations at a fraction of the cost and time. Consistent with prior work&#x2014;including validation by Zheng et al [<xref ref-type="bibr" rid="ref31">31</xref>] in MT-Bench and Chatbot Arena, which showed that structured prompts and rubric-based scoring enabled LLM referees to approximate expert-level consistency&#x2014;our results further support the feasibility of automated evaluation frameworks. The 2 referee models (Qwen3-Max and DeepSeek-R1) showed moderate to good agreement across rounds, indicating that standardized rubrics and controlled prompting can reduce evaluation variance. However, the lower stability of DeepSeek-R1 compared with Qwen3-Max may reflect model-specific differences in prompt sensitivity, reasoning behavior, and evidence integration; as a reasoning-oriented model, DeepSeek-R1 may be more affected by prompt wording, contextual structure, and the ordering or redundancy of retrieved evidence fragments, whereas Qwen3-Max may have followed the scoring rubric and formatted output more consistently [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. These findings extend previous evidence and reinforce the role of automated referees as a practical, scalable tool for benchmarking domain-specific LLMs and guiding iterative model improvement.</p><p>In our expert-LLM consistency analysis, ICC values were in the moderate range, indicating that although the overall ranking trends across models were aligned, finer-grained scoring differed. This pattern is consistent with recent literature suggesting that human experts and LLM judges operate with distinct evaluative priors [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. Human experts typically emphasize clinical risk, actionability, contextual relevance, and patient-specific feasibility [<xref ref-type="bibr" rid="ref35">35</xref>]&#x2014;dimensions inherently subjective and embedded in real-world clinical reasoning. In contrast, LLM referees tend to weight structured coherence, completeness, and evidence alignment more heavily.</p><p>Therefore, the observed discrepancy does not reflect disagreement in model performance but rather divergence in evaluation emphasis. These differences underscore that human and automated assessments are complementary rather than interchangeable. In this study, LLM referees were positioned as scalable, standardized, and reproducible tools for rubric-based evaluation rather than as substitutes for human clinical experts. They can efficiently assess evidence alignment, comprehensibility, relevance, empathy, and feasibility under unified criteria [<xref ref-type="bibr" rid="ref36">36</xref>], but their scoring may still be influenced by prompt design, model-specific preferences, training corpora, and task framing. In contrast, human experts provide experience-based judgment on patient safety, individualized applicability [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>], risk boundaries, and practical feasibility [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. Integrating both perspectives enables a more comprehensive evaluation framework and mitigates the limitations associated with relying on either method.</p></sec><sec id="s4-2"><title>Strengths and Limitations</title><p>This study offers several methodological innovations beyond conventional SFT- or RAG-based biomedical LLM development: (1) Automated literature-to-QA generation pipeline: we proposed a fully automated workflow that transforms high-quality evidence into structured QA pairs, enabling efficient, standardized, and scalable construction of domain-specific training corpora. (2) A generalizable methodological paradigm for chronic disease management: the modular framework&#x2014;integrating knowledge slicing, structured QA generation, and domain-adaptive SFT&#x2014;can be readily extended to other chronic disease contexts, providing a reusable blueprint for developing patient education LLMs. (3) Evidence-grounded evaluation framework using a RAG-augmented referee LLM: we established a structured referee LLM&#x2013;based multidimensional assessment system, enabling standardized evaluation of evidence alignment, safety, clarity, and relevance.</p><p>This study has several limitations. First, although the model&#x2019;s responses integrate evidence-based medical knowledge, their effectiveness and safety have not been validated in real patient interactions; this remains a major limitation for a patient-facing system and an important prerequisite for clinical translation. In addition, the manual QA audit underscores the need for upstream quality control in automated literature-to-QA generation. Although revised and filtered samples were used for SFT, residual corpus noise may remain. Despite the additional interrater agreement analysis, full-corpus manual validation was not performed. Therefore, PCaPLMM_SFT should currently be regarded as a proof-of-concept and a patient education support tool rather than a stand-alone clinically deployable system. Second, despite using multiple referee models and incorporating human expert assessments, bias related to model architecture, training corpora, or corpus-generation procedures may still persist. Although the referee LLM outputs were examined through interround, intermodel, and human-LLM consistency analyses, the referee scores should be interpreted as auxiliary evaluation results rather than definitive expert adjudications. In particular, alternative QA generators were not systematically tested; therefore, potential generator-induced bias from Qwen3-Turbo may remain. Future studies should compare multiple QA generators and incorporate cross-model review or expert-annotated reference sets to further strengthen corpus diversity, evaluation robustness, and generalizability. Third, the human expert validation was based on a limited sample of 50 QA pairs across 5 lifestyle scenarios. Therefore, the expert evaluation should be interpreted as a supplementary consistency check rather than a fully representative assessment of the entire QA corpus or broader patient-facing use scenarios. Finally, as this work primarily focused on methodological development, real-world behavioral outcomes and long-term follow-up effects could not yet be assessed.</p><p>Future work should expand real-world interaction datasets and incorporate patient feedback to enhance external validity. Methodologically, applying stronger regularization, early stopping, and multisource corpora may further improve model generalization. In addition, the failure cases in <xref ref-type="table" rid="table5">Table 5</xref> indicate that hallucination risks, including fabricated or irrelevant citations, cannot be fully eliminated by RAG and manual review alone; future work could integrate automated fact-checking modules for claim decomposition, citation verification, and evidence support assessment, while retaining expert review in clinically sensitive scenarios. Given the variability observed across referee LLMs and human experts, establishing a multilayer evaluation system that integrates LLM judges, clinicians, and patient users will be essential for improving the stability and safety assessment of model behavior. Finally, extending the proposed literature-to-QA pipeline and structured evaluation paradigm to other diseases represents an important direction for developing scalable, evidence-grounded patient education LLMs. Further validation with broader Chinese-language clinical and patient education resources is also warranted. Moreover, real-world RAG deployment involving multiple retrieved fragments and multiturn dialogue history may require adaptive chunk selection, history summarization, or longer-context models beyond the context window settings used in this study.</p></sec><sec id="s4-3"><title>Conclusions</title><p>As an SFT LLM developed for healthy lifestyle management in patients with prostate cancer, PCaPLMM_SFT supports the feasibility of combining domain-specific knowledge injection with structured QA-based training strategies. The model generated evidence-supported, comprehensible, and empathetic responses across lifestyle-related dialogue scenarios, suggesting its potential as a research prototype and patient education support tool under appropriate clinical oversight. This work also establishes a feasible and reproducible framework for developing disease-specific generative dialogue systems by integrating automated literature-to-QA construction and dual-layer evaluation. The methodology may provide guidance for adapting similar models to other chronic disease contexts. Moving forward, real-world patient interaction studies, iterative dataset refinement, and multistakeholder human-AI coevaluation will be needed before such systems can be considered for broader patient-facing deployment.</p></sec></sec></body><back><ack><p>The authors sincerely thank the 3 experts who participated in the evaluation and validation of this model, as well as the graduate students who contributed to the data collection and organization. Generative AI tools were used only to assist with language polishing and clarity of expression. No generative AI was used for study design, data collection, data analysis, statistical modeling, algorithm development, figure or table generation, or interpretation of results. All AI-assisted text was reviewed, edited, and verified by the authors, who take full responsibility for the accuracy, integrity, and final content of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the National Natural Science Foundation of China (grants 82102186 and 32200533). The funder played no role in the study design, data collection, analysis, and interpretation of data, or the writing of this manuscript.</p></sec><sec><title>Data Availability</title><p>The model weights and configuration files generated during this study are available in the Hugging Face repository [<xref ref-type="bibr" rid="ref41">41</xref>]. The repository includes the final model weights, training configurations, and documentation necessary to reproduce the reported results. No individual participant data were collected or shared.</p></sec></notes><fn-group><fn fn-type="con"><p>KJ and YC conceived and designed the study, supervised the overall research process, and administered the project. FJ developed the methodology, implemented the software, curated the dataset, and performed the formal analyses together with YL. QY, XZ, NY, and JZ contributed to data curation, visualization, and investigation, supporting the analytical workflow led by FJ and YL. TL and YW assisted in data acquisition and preliminary data management. FJ and QY were responsible for training quality control. FJ drafted the original manuscript. KJ and YC, together with FJ, critically reviewed and revised the manuscript. All authors had full access to the data presented in the study, contributed to the interpretation of the findings, and approved the final version of the manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ADT</term><def><p>androgen deprivation therapy</p></def></def-item><def-item><term id="abb2">FAISS</term><def><p>Facebook Artificial Intelligence Similarity Search</p></def></def-item><def-item><term id="abb3">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb4">KB</term><def><p>knowledge base</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">PCaPLMM_SFT</term><def><p>Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-tuning</p></def></def-item><def-item><term id="abb7">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb8">QA</term><def><p>question-answer</p></def></def-item><def-item><term id="abb9">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb10">SFT</term><def><p>supervised fine-tuning</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>PCaLiStDB: a lifestyle database for precision prevention of prostate cancer</article-title><source>Database (Oxford)</source><year>2020</year><month>01</month><day>1</day><volume>2020</volume><fpage>baz154</fpage><pub-id pub-id-type="doi">10.1093/database/baz154</pub-id><pub-id pub-id-type="medline">31950190</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>K</given-names> </name><name name-style="western"><surname>Park</surname><given-names>J</given-names> </name><name name-style="western"><surname>Oh</surname><given-names>EG</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Park</surname><given-names>C</given-names> </name><name name-style="western"><surname>Choi</surname><given-names>YD</given-names> </name></person-group><article-title>Effectiveness of a nurse-led mobile-based health coaching program for patients with prostate cancer at high risk of metabolic syndrome: randomized waitlist controlled trial</article-title><source>JMIR Mhealth Uhealth</source><year>2024</year><month>02</month><day>1</day><volume>12</volume><fpage>e47102</fpage><pub-id pub-id-type="doi">10.2196/47102</pub-id><pub-id pub-id-type="medline">38300697</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wright</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Schenk</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Gulati</surname><given-names>R</given-names> </name><etal/></person-group><article-title>The Prostate Cancer Active Lifestyle Study (PALS): a randomized controlled trial of diet and exercise in overweight and obese men on active surveillance</article-title><source>Cancer</source><year>2024</year><month>06</month><day>15</day><volume>130</volume><issue>12</issue><fpage>2108</fpage><lpage>2119</lpage><pub-id pub-id-type="doi">10.1002/cncr.35241</pub-id><pub-id pub-id-type="medline">38353455</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McNaught</surname><given-names>E</given-names> </name><name name-style="western"><surname>Reale</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bourke</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Supported exercise TrAining for Men wIth prostate caNcer on Androgen deprivation therapy (STAMINA): study protocol for a randomised controlled trial of the clinical and cost-effectiveness of the STAMINA lifestyle intervention compared with optimised usual care, including internal pilot and parallel process evaluation</article-title><source>Trials</source><year>2024</year><month>04</month><day>12</day><volume>25</volume><issue>1</issue><fpage>257</fpage><pub-id pub-id-type="doi">10.1186/s13063-024-07989-y</pub-id><pub-id pub-id-type="medline">38610058</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harris</surname><given-names>E</given-names> </name></person-group><article-title>Large language models answer medical questions accurately, but can&#x2019;t match clinicians&#x2019; knowledge</article-title><source>JAMA</source><year>2023</year><month>09</month><day>5</day><volume>330</volume><issue>9</issue><fpage>792</fpage><lpage>794</lpage><pub-id pub-id-type="doi">10.1001/jama.2023.14311</pub-id><pub-id pub-id-type="medline">37548971</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Long</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>M</given-names> </name><etal/></person-group><article-title>BioRAG: a RAG-LLM framework for biological question reasoning</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 14, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2408.01107</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chow</surname><given-names>JCL</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name></person-group><article-title>Large language models in medical chatbots: opportunities, challenges, and the need to address AI risks</article-title><source>Information</source><year>2025</year><month>07</month><volume>16</volume><issue>7</issue><fpage>549</fpage><pub-id pub-id-type="doi">10.3390/info16070549</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chow</surname><given-names>JCL</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name></person-group><article-title>Developing effective frameworks for large language model-based medical chatbots: insights from radiotherapy education with ChatGPT</article-title><source>JMIR Cancer</source><year>2025</year><month>02</month><day>18</day><volume>11</volume><issue>1</issue><fpage>e66633</fpage><pub-id pub-id-type="doi">10.2196/66633</pub-id><pub-id pub-id-type="medline">39965195</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><etal/></person-group><article-title>EyeGPT for patient inquiries and medical education: development and validation of an ophthalmology large language model</article-title><source>J Med Internet Res</source><year>2024</year><month>12</month><day>11</day><volume>26</volume><issue>1</issue><fpage>e60063</fpage><pub-id pub-id-type="doi">10.2196/60063</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li&#x00E9;vin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Hother</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Motzfeldt</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Winther</surname><given-names>O</given-names> </name></person-group><article-title>Can large language models reason about medical questions?</article-title><source>Patterns (N Y)</source><year>2024</year><month>03</month><day>8</day><volume>5</volume><issue>3</issue><fpage>100943</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2024.100943</pub-id><pub-id pub-id-type="medline">38487804</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sierra</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>B</given-names> </name></person-group><article-title>Predictive model for daily risk alerts in sepsis patients in the ICU: visualization and clinical analysis of risk indicators</article-title><source>Precis Clin Med</source><year>2025</year><month>03</month><volume>8</volume><issue>1</issue><fpage>baf003</fpage><pub-id pub-id-type="doi">10.1093/pcmedi/pbaf003</pub-id><pub-id pub-id-type="medline">40041421</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><etal/></person-group><article-title>A knowledge-enhanced platform (MetaSepsisKnowHub) for retrieval augmented generation-based sepsis heterogeneity and personalized management: development study</article-title><source>J Med Internet Res</source><year>2025</year><month>06</month><day>6</day><volume>27</volume><fpage>e67201</fpage><pub-id pub-id-type="doi">10.2196/67201</pub-id><pub-id pub-id-type="medline">40478618</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>He</surname><given-names>M</given-names> </name><name name-style="western"><surname>Feng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>B</given-names> </name></person-group><article-title>Graph retrieval augmented large language models for facial phenotype associated rare genetic disease</article-title><source>NPJ Digit Med</source><year>2025</year><month>08</month><day>24</day><volume>8</volume><issue>1</issue><fpage>543</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01955-x</pub-id><pub-id pub-id-type="medline">40849403</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>B</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Baichuan 2: open large-scale language models</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 17, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2309.10305</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Izacard</surname><given-names>G</given-names> </name><name name-style="western"><surname>Grave</surname><given-names>E</given-names> </name></person-group><article-title>Leveraging passage retrieval with generative models for open domain question answering</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 3, 2021</comment><pub-id pub-id-type="doi">10.48550/arXiv.2007.01282</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>NF</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hewitt</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Lost in the middle: how language models use long contexts</article-title><source>Trans Assoc Comput Linguist</source><year>2024</year><month>02</month><day>23</day><volume>12</volume><fpage>157</fpage><lpage>173</lpage><pub-id pub-id-type="doi">10.1162/tacl_a_00638</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amugongo</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Mascheroni</surname><given-names>P</given-names> </name><name name-style="western"><surname>Brooks</surname><given-names>S</given-names> </name><name name-style="western"><surname>Doering</surname><given-names>S</given-names> </name><name name-style="western"><surname>Seidel</surname><given-names>J</given-names> </name></person-group><article-title>Retrieval augmented generation for large language models in healthcare: a systematic review</article-title><source>PLoS Digit Health</source><year>2025</year><month>06</month><volume>4</volume><issue>6</issue><fpage>e0000877</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000877</pub-id><pub-id pub-id-type="medline">40498738</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ning</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Keppo</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation for generative artificial intelligence in health care</article-title><source>NPJ Health Syst</source><year>2025</year><month>01</month><day>25</day><volume>2</volume><issue>1</issue><fpage>2</fpage><pub-id pub-id-type="doi">10.1038/s44401-024-00004-1</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kordi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Mishra</surname><given-names>S</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Rogers</surname><given-names>A</given-names> </name><name name-style="western"><surname>Boyd-Graber</surname><given-names>J</given-names> </name><name name-style="western"><surname>Okazaki</surname><given-names>N</given-names> </name></person-group><article-title>Self-instruct: aligning language models with self-generated instructions</article-title><year>2023</year><conf-name>Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>Jul 9-14, 2023</conf-date><conf-loc>Toronto, Canada</conf-loc><publisher-name>Association for Computational Linguistics</publisher-name><fpage>13484</fpage><lpage>13508</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.acl-long.754</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fast</surname><given-names>D</given-names> </name><name name-style="western"><surname>Adams</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Autonomous medical evaluation for guideline adherence of large language models</article-title><source>NPJ Digit Med</source><year>2024</year><month>12</month><day>12</day><volume>7</volume><issue>1</issue><fpage>358</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01356-6</pub-id><pub-id pub-id-type="medline">39668168</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Croxford</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>First</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Evaluating clinical AI summaries with large language models as judges</article-title><source>NPJ Digit Med</source><year>2025</year><month>11</month><day>5</day><volume>8</volume><issue>1</issue><fpage>640</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-02005-2</pub-id><pub-id pub-id-type="medline">41193667</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gaebe</surname><given-names>K</given-names> </name><name name-style="western"><surname>van der Woerd</surname><given-names>B</given-names> </name></person-group><article-title>Evaluation of large language models as a diagnostic tool for medical learners and clinicians using advanced prompting techniques</article-title><source>PLoS One</source><year>2025</year><volume>20</volume><issue>8</issue><fpage>e0325803</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0325803</pub-id><pub-id pub-id-type="medline">40749008</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Azizi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language models encode clinical knowledge</article-title><source>Nature</source><year>2023</year><month>08</month><volume>620</volume><issue>7972</issue><fpage>172</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id><pub-id pub-id-type="medline">37438534</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dennst&#x00E4;dt</surname><given-names>F</given-names> </name><name name-style="western"><surname>Schmerder</surname><given-names>M</given-names> </name><name name-style="western"><surname>Riggenbach</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Comparative evaluation of a medical large language model in answering real-world radiation oncology questions: multicenter observational study</article-title><source>J Med Internet Res</source><year>2025</year><month>09</month><day>23</day><volume>27</volume><issue>1</issue><fpage>e69752</fpage><pub-id pub-id-type="doi">10.2196/69752</pub-id><pub-id pub-id-type="medline">40986858</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tam</surname><given-names>TYC</given-names> </name><name name-style="western"><surname>Sivarajkumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kapoor</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A framework for human evaluation of large language models in healthcare derived from literature review</article-title><source>NPJ Digit Med</source><year>2024</year><month>09</month><day>28</day><volume>7</volume><issue>1</issue><fpage>258</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01258-7</pub-id><pub-id pub-id-type="medline">39333376</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>F</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Bouamor</surname><given-names>H</given-names> </name><name name-style="western"><surname>Pino</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bali</surname><given-names>K</given-names> </name></person-group><article-title>HuatuoGPT, towards taming language model to be a doctor</article-title><source>Findings of the Association for Computational Linguistics: EMNLP 2023</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>10859</fpage><lpage>10885</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.findings-emnlp.725</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>P</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>RadOnc-GPT: a large language model for radiation oncology</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 6, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2309.10160</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baysan</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Uysal</surname><given-names>S</given-names> </name><name name-style="western"><surname>&#x0130;&#x015F;lek</surname><given-names>&#x0130;</given-names> </name><name name-style="western"><surname>&#x00C7;&#x0131;&#x011F; Karaman</surname><given-names>&#x00C7;</given-names> </name><name name-style="western"><surname>G&#x00FC;ng&#x00F6;r</surname><given-names>T</given-names> </name></person-group><article-title>LLM-as-a-Judge: automated evaluation of search query parsing using large language models</article-title><source>Front Big Data</source><year>2025</year><volume>8</volume><fpage>1611389</fpage><pub-id pub-id-type="doi">10.3389/fdata.2025.1611389</pub-id><pub-id pub-id-type="medline">40761620</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>JudgeLRM: large reasoning models as a judge</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 3, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.00050</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>D</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>L</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Christodoulopoulos</surname><given-names>C</given-names> </name><name name-style="western"><surname>Chakraborty</surname><given-names>T</given-names> </name><name name-style="western"><surname>Rose</surname><given-names>C</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>V</given-names> </name></person-group><article-title>From generation to judgment: opportunities and challenges of LLM-as-a-judge</article-title><source>The 2025 Conference on Empirical Methods in Natural Language Processing</source><year>2025</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>2757</fpage><lpage>2791</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.emnlp-main.138</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zheng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chiang</surname><given-names>WL</given-names> </name><name name-style="western"><surname>Sheng</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Judging LLM-as-a-Judge with MT-Bench and chatbot arena</article-title><source>arXiv</source><year>2023</year><month>12</month><day>24</day><pub-id pub-id-type="doi">10.48550/arXiv.2306.05685</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>G</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>B</given-names> </name><etal/></person-group><article-title>A multi-dimensional performance evaluation of large language models in dental implantology: comparison of ChatGPT, DeepSeek, Grok, Gemini and Qwen across diverse clinical scenarios</article-title><source>BMC Oral Health</source><year>2025</year><month>07</month><day>28</day><volume>25</volume><issue>1</issue><fpage>1272</fpage><pub-id pub-id-type="doi">10.1186/s12903-025-06619-6</pub-id><pub-id pub-id-type="medline">40721763</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baur</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ansorg</surname><given-names>J</given-names> </name><name name-style="western"><surname>Heyde</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Voelker</surname><given-names>A</given-names> </name></person-group><article-title>Development and evaluation of a retrieval-augmented generation chatbot for orthopedic and trauma surgery patient education: mixed-methods study</article-title><source>JMIR AI</source><year>2025</year><month>10</month><day>23</day><volume>4</volume><issue>1</issue><fpage>e75262</fpage><pub-id pub-id-type="doi">10.2196/75262</pub-id><pub-id pub-id-type="medline">41134117</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Haldar</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hockenmaier</surname><given-names>J</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Christodoulopoulos</surname><given-names>C</given-names> </name><name name-style="western"><surname>Chakraborty</surname><given-names>T</given-names> </name><name name-style="western"><surname>Rose</surname><given-names>C</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>V</given-names> </name></person-group><article-title>Rating roulette: self-inconsistency in LLM-as-a-judge frameworks</article-title><source>Findings of the Association for Computational Linguistics</source><year>2025</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>24986</fpage><lpage>25004</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.findings-emnlp.1361</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gottweis</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="medline">39779926</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>X</given-names> </name><name name-style="western"><surname>Ruan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>X</given-names> </name></person-group><article-title>LLM-based NLG evaluation: current status and challenges</article-title><source>Comput Linguist</source><year>2025</year><month>06</month><day>24</day><volume>51</volume><issue>2</issue><fpage>661</fpage><lpage>687</lpage><pub-id pub-id-type="doi">10.1162/coli_a_00561</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Montori</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Guyatt</surname><given-names>GH</given-names> </name></person-group><article-title>What is evidence-based medicine?</article-title><source>Endocrinol Metab Clin North Am</source><year>2002</year><month>09</month><volume>31</volume><issue>3</issue><fpage>521</fpage><lpage>526</lpage><pub-id pub-id-type="doi">10.1016/s0889-8529(02)00015-4</pub-id><pub-id pub-id-type="medline">12227116</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Collins</surname><given-names>J</given-names> </name></person-group><article-title>Evidence-based medicine</article-title><source>J Am Coll Radiol</source><year>2007</year><month>08</month><volume>4</volume><issue>8</issue><fpage>551</fpage><lpage>554</lpage><pub-id pub-id-type="doi">10.1016/j.jacr.2006.12.007</pub-id><pub-id pub-id-type="medline">17660119</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Choudhury</surname><given-names>A</given-names> </name><name name-style="western"><surname>Asan</surname><given-names>O</given-names> </name></person-group><article-title>Role of artificial intelligence in patient safety outcomes: systematic literature review</article-title><source>JMIR Med Inform</source><year>2020</year><month>07</month><day>24</day><volume>8</volume><issue>7</issue><fpage>e18599</fpage><pub-id pub-id-type="doi">10.2196/18599</pub-id><pub-id pub-id-type="medline">32706688</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>JYB</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>XL</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>I</given-names> </name></person-group><article-title>Barriers and enablers to implementing clinical practice guidelines in primary care: an overview of systematic reviews</article-title><source>BMJ Open</source><year>2023</year><month>01</month><day>6</day><volume>13</volume><issue>1</issue><fpage>e062158</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2022-062158</pub-id><pub-id pub-id-type="medline">36609329</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="web"><article-title>PCaPLMM_SFT: a prostate cancer patient lifestyle management model via supervised fine-tuning</article-title><source>Hugging Face</source><access-date>2026-07-06</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/RomilY/PCaPLMM_SFT">https://huggingface.co/RomilY/PCaPLMM_SFT</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>PubMed search strategy and inclusion and exclusion.</p><media xlink:href="jmir_v28i1e92663_app1.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Detailed manual quality control evaluation results for sampled question-answer pairs.</p><media xlink:href="jmir_v28i1e92663_app2.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Prompt templates for the large language model referees.</p><media xlink:href="jmir_v28i1e92663_app3.docx" xlink:title="DOCX File, 19 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Five-dimension scoring framework.</p><media xlink:href="jmir_v28i1e92663_app4.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Expert evaluation form for multiple model outputs on lifestyle management in patients with prostate cancer.</p><media xlink:href="jmir_v28i1e92663_app5.docx" xlink:title="DOCX File, 29 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6</label><p>Classification of manual quality control outcomes for the sampled question-answer pairs.</p><media xlink:href="jmir_v28i1e92663_app6.docx" xlink:title="DOCX File, 15 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7</label><p>Representative failure cases during manual quality control.</p><media xlink:href="jmir_v28i1e92663_app7.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app8"><label>Multimedia Appendix 8</label><p>Proportion of different error types across models.</p><media xlink:href="jmir_v28i1e92663_app8.docx" xlink:title="DOCX File, 17 KB"/></supplementary-material><supplementary-material id="app9"><label>Multimedia Appendix 9</label><p>Human expert evaluation scores for candidate models.</p><media xlink:href="jmir_v28i1e92663_app9.docx" xlink:title="DOCX File, 19 KB"/></supplementary-material><supplementary-material id="app10"><label>Multimedia Appendix 10</label><p>Full responses of the example question-answer pairs generated by PCaPLMM_SFT (Prostate Cancer Patient Lifestyle Management Model via Supervised Fine-Tuning).</p><media xlink:href="jmir_v28i1e92663_app10.docx" xlink:title="DOCX File, 22 KB"/></supplementary-material></app-group></back></article>