<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e99136</article-id><article-id pub-id-type="doi">10.2196/99136</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>LLM-Generated Lay-Language Protocols for Molecular Tumor Board Patients: Evaluation of Quality and Clinical Usability</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Pakull</surname><given-names>Tabea Margareta Grace</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Bender</surname><given-names>No&#x00EB;lle</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Benson</surname><given-names>Sven</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Fleischhauer</surname><given-names>Anke</given-names></name><degrees>BA</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Alsara</surname><given-names>Mohammad</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gromke</surname><given-names>Tanja</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hilser</surname><given-names>Thomas</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kaminski</surname><given-names>Katharina</given-names></name><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pogorzelski</surname><given-names>Michael</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Prasuhn</surname><given-names>Nicola</given-names></name><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Rosery</surname><given-names>Vivian</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Schadendorf</surname><given-names>Dirk</given-names></name><degrees>Prof Dr med</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="aff" rid="aff9">9</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Schuler</surname><given-names>Martin</given-names></name><degrees>Prof Dr med</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wiesweg</surname><given-names>Marcel</given-names></name><degrees>Prof Dr med</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zaun</surname><given-names>Gregor</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Horn</surname><given-names>Peter Alexander</given-names></name><degrees>Prof Dr med</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Friedrich</surname><given-names>Christoph Matthias</given-names></name><degrees>Prof Dr-Ing</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff10">10</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pretzell</surname><given-names>Ina</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib></contrib-group><aff id="aff1"><institution>Institute for Transfusion Medicine, Essen University Hospital</institution><addr-line>Hufelandstra&#x00DF;e 55</addr-line><addr-line>Essen</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff2"><institution>Department of Computer Science, Dortmund University of Applied Sciences and Arts</institution><addr-line>Dortmund</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff3"><institution>Department of Human-Centered Computing and Cognitive Science, Social Psychology, University of Duisburg-Essen</institution><addr-line>Duisburg</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff4"><institution>Centre for Translational Neuro- and Behavioral Sciences (C-TNBS), Institute for Medical Education, Essen University Hospital</institution><addr-line>Essen</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff5"><institution>West German Cancer Center (WTZ), Essen University Hospital</institution><addr-line>Essen</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff6"><institution>Department of Medical Oncology, West German Cancer Center (WTZ), Essen University Hospital</institution><addr-line>Essen</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff7"><institution>Patient Partner, Patient Advisory Board, West German Cancer Center (WTZ), Essen University Hospital</institution><addr-line>Essen</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff8"><institution>NCT West, National Center for Tumor Diseases (NCT)</institution><addr-line>Essen</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff9"><institution>Department of Dermatology, West German Cancer Center (WTZ), Essen University Hospital</institution><addr-line>Essen</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff10"><institution>Institute for Medical Informatics, Biometry and Epidemiology (IMIBE), Essen University Hospital</institution><addr-line>Essen</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Yadav</surname><given-names>Vijay</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Hu</surname><given-names>Yihan</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Ding</surname><given-names>Zhanyi</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Liu</surname><given-names>Zhao</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Tabea Margareta Grace Pakull, MSc, Institute for Transfusion Medicine, Essen University Hospital, Hufelandstra&#x00DF;e 55, Essen, North Rhine-Westphalia, 45147, Germany, 49 2017234201; <email>tabea.pakull@uk-essen.de</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>23</day><month>7</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e99136</elocation-id><history><date date-type="received"><day>22</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>22</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>23</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Tabea Margareta Grace Pakull, No&#x00EB;lle Bender, Sven Benson, Anke Fleischhauer, Mohammad Alsara, Tanja Gromke, Thomas Hilser, Katharina Kaminski, Michael Pogorzelski, Nicola Prasuhn, Vivian Rosery, Dirk Schadendorf, Martin Schuler, Marcel Wiesweg, Gregor Zaun, Peter Alexander Horn, Christoph Matthias Friedrich, Ina Pretzell. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 23.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e99136"/><abstract><sec><title>Background</title><p>Molecular Tumor Boards (MTBs) generate highly technical recommendations. The language used in their protocols is rarely accessible to patients. Lay-language patient protocols could support patient-clinician communication, yet manual production is difficult to sustain in high-volume oncology settings. Large language models (LLMs) may offer scalable drafting assistance, yet clinical usability remains largely uninvestigated under real-world deployment constraints. Existing evaluations rely predominantly on synthetic data or closed-source models that are incompatible with strict data protection requirements.</p></sec><sec><title>Objective</title><p>This study evaluated whether open-weight LLMs can provide clinically usable drafting support for German MTB patient protocols under real-world deployment constraints and developed a transferable evaluation framework for patient-facing text generation.</p></sec><sec sec-type="methods"><title>Methods</title><p>Eight open-weight LLMs were evaluated under zero-shot (A1) and one-shot (A2) prompting with constrained decoding, which ensures section-schema compliance. Automatic evaluation used ROUGE-1 (Recall-Oriented Understudy for Gisting Evaluation), BERTScore-F1 (Bidirectional Encoder Representations From Transformers Score), Wiener Sachtextformel version 4, and DistilBERT (Distilled Version of Bidirectional Encoder Representations From Transformers)&#x2013;based complexity using a corpus of 316 MTB protocols and 47 expert-written patient protocols. For expert evaluation, 7 medical oncologists evaluated 50 protocols from the best-performing model across 3 International Organization for Standardization 9241&#x2010;11 usability dimensions using fine-grained error annotation, perceived postediting effort (PPEE), and net promoter score. Critical errors were defined as bearing the risk of patient harm.</p></sec><sec sec-type="results"><title>Results</title><p>Llama-3.3-70B-Instruct achieved the strongest automatic performance. Across models, A2 significantly improved most automatic metrics compared to A1. However, expert usability evaluation of Llama-3.3-70B-Instruct showed the opposite picture: the proportion of protocols containing at least 1 critical error doubled under A2 (10/25, 40% vs 5/25, 20%) compared with A1, and the dominant error type shifted from language (40/108, 37%) errors to factual errors (69/145, 48%). Overall, 16% (230/1420) of the annotated paragraphs contained errors. Median PPEE was 2 (IQR 2.0&#x2010;3.0; low), and median net promoter score was 7 (IQR 5.0-9.0). Detractors (46/100, 46%) outweighed promoters (29/100, 29%), which suggests hesitation toward routine adoption. These differences in expert evaluation between A2 and A1 were directionally consistent but did not reach individual statistical significance for the paired samples (n=25).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Prompting strategies that improve automatic metrics can simultaneously increase the number of critical errors. Surface-level metric gains were, therefore, insufficient proxies for clinical safety. This was observed as a consistent directional pattern for a single model, but generalization to other models remains to be investigated. Nonetheless, the low paragraph-level error rate and favorable PPEE suggest that structured open-weight LLM generation may be a useful drafting support in a clinician-supervised setting. The proposed evaluation framework provides a text-quality-focused basis for future assessment of patient-facing LLM applications in real-world clinical settings.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>natural language processing</kwd><kwd>artificial intelligence</kwd><kwd>AI</kwd><kwd>precision medicine</kwd><kwd>lay language</kwd><kwd>Molecular Tumor Board</kwd><kwd>German</kwd><kwd>health literacy</kwd><kwd>evaluation</kwd><kwd>usability</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Effective clinician-patient communication is fundamental to patient-centered care and shared decision-making [<xref ref-type="bibr" rid="ref1">1</xref>], with demonstrated benefits for patient understanding, treatment adherence, and clinical outcomes [<xref ref-type="bibr" rid="ref2">2</xref>]. This challenge is particularly acute in oncology. Molecular Tumor Boards (MTBs) [<xref ref-type="bibr" rid="ref3">3</xref>] synthesize complex genomic and molecular findings to formulate individualized therapy recommendations for patients with advanced malignancies. Their recommendations are depicted in highly technical protocols that are largely inaccessible to lay readers. Written, lay-language summaries of MTB outcomes could support patient participation in treatment decisions and improve transparency in precision oncology care, yet their manual preparation is time-consuming and difficult to sustain in high-volume clinical environments. Automated drafting assistance for oncologists, as the intended users of such systems, is therefore a practically relevant prospect.</p><p>Large language models (LLMs) have attracted considerable interest for clinical communication tasks [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. Their capacity for instruction-following and fluent text generation with minimal task-specific adaptation [<xref ref-type="bibr" rid="ref6">6</xref>] positions them as candidate tools for lay-language protocol drafting. However, integration into clinical workflows remains constrained by well-documented concerns regarding factual accuracy, potential for harm, and output bias [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. In oncology, clinician oversight of generated content is therefore structurally required rather than optional.</p><p>Two additional constraints narrow the practical design space. First, most published evaluations have focused on closed-source models [<xref ref-type="bibr" rid="ref5">5</xref>] such as GPT-4 (OpenAI) [<xref ref-type="bibr" rid="ref8">8</xref>]. These models are difficult to deploy for real patient data under current German and European data protection requirements. Therefore, open-weight alternatives that support on-premise deployment are the only viable option for most settings. Second, most existing studies were conducted on synthetic or deidentified data [<xref ref-type="bibr" rid="ref4">4</xref>] and therefore offer limited relevance to operational clinical practice. Evaluation on real patient data under realistic hardware constraints remains an exception.</p><p>Another practical requirement is output conformity to institutional standards. Structured generation [<xref ref-type="bibr" rid="ref9">9</xref>] addresses this by constraining fixed structural elements via predefined schemas and has been shown to improve format adherence and reduce hallucination rates in clinical note generation [<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>Our preliminary work [<xref ref-type="bibr" rid="ref11">11</xref>] evaluated LLM-generated patient protocols from 4 MTB protocols using a single open-weight model with section-wise prompting. This approach led to narratively incoherent outputs across sections. Additionally, an LLM-as-a-judge approach proved insufficient to distinguish safe simplifications from factual errors, and a single-reviewer assessment without clinical expertise is methodologically inadequate for quality assurance in this setting. These findings identified 3 methodological requirements for a more rigorous follow-up study: structured generation for schema compliance, multimodel evaluation under realistic deployment constraints, and expert-led evaluation using a clinically grounded error taxonomy. The central challenge in this setting was not merely the generation of fluent lay-language communication, but to determine whether such outputs are clinically usable and safe as clinician-supervised drafting assistance.</p><p>This study addresses these requirements through 3 contributions. An overview of this study&#x2019;s design is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. First, it introduces a transferable multilevel evaluation framework operationalizing clinical usability across the 3 International Organization for Standardization (ISO) 9241&#x2010;11 [<xref ref-type="bibr" rid="ref12">12</xref>] dimensions of effectiveness, efficiency, and satisfaction. Second, it evaluates a diverse panel of 8 open-weight LLMs under zero-shot and style-conditioned one-shot prompting. Third, it characterizes the relationship between prompting strategy, automatic metric performance, and expert-assessed clinical quality, and examines whether improvements in automatic metrics correspond to expert-assessed clinical quality. Taken together, these contributions serve 2 aims: introducing a clinical usability evaluation framework for lay language generation and determining the clinical usability of LLM-generated patient protocols under real-world deployment constraints.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study design overview showing data sources (MTB protocols and expert-written patient protocols), LLM-based structured generation under zero-shot (A1) and style-conditioned one-shot (A2) prompting, and a multidimensional evaluation framework assessing automatic metrics and clinical usability. BERTScore: Bidirectional Encoder Representations From Transformers Score; DistilBERT: Distilled Version of Bidirectional Encoder Representations From Transformers; LLM: large language model; MTB: Molecular Tumor Board; ROUGE: Recall-Oriented Understudy for Gisting Evaluation; WSTF4: Wiener Sachtextformel version 4.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e99136_fig01.png"/></fig></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Data</title><p>Two German corpora of MTB and patient protocols were used in this study. <xref ref-type="fig" rid="figure2">Figure 2</xref> shows the corpus construction process, and <xref ref-type="table" rid="table1">Table 1</xref> summarizes its statistics.</p><p>A parallel corpus of 47 MTB protocols and expert-written patient protocols served as the gold standard for evaluation. The patient protocols were authored by an experienced oncologist specialized in precision oncology who leads an institutional MTB at the West German Cancer Center (WTZ) in Essen, Germany. The language and structure of the protocols align with a guideline developed by communication specialists, medical didacts, and the patient advisory board of the WTZ to ensure audience-appropriate language. A subset of the patient protocols (n=33) was evaluated in the MyCODE study [<xref ref-type="bibr" rid="ref13">13</xref>], which supports their suitability as gold-standard references through positive assessments by lay readers and clinicians.</p><p>Protocols were categorized into protocols with therapy recommendation (n=22), where actionable molecular findings supported targeted therapy recommendations, and protocols without therapy recommendation (n=25), where no actionable molecular alterations were identified. These 2 case types follow distinct section schemas. Protocols with therapy recommendations were longer, as they contained more sections than protocols without therapy recommendations, primarily due to additional explanations of individual findings and supporting literature.</p><p>Patient protocols were manually segmented into 410 sections (227 with therapy recommendation and 183 without therapy recommendation) for reference-based evaluation.</p><p>For large-scale evaluation, 607 MTB protocols were exported via Fast Healthcare Interoperability Resources export from the smart hospital information platform, restricted to patients with Medical Informatics Initiative broad consent issued after January 1, 2024, and before September 25, 2025. After deduplication to remove overlaps with the parallel corpus, 562 unique MTB protocols were retained. A further 293 protocols were removed because they were not intended for patient communication. These documents only record an internal request for further diagnostic work and contain no findings or therapy recommendations to convey to patients. Finally, 269 protocols remained (with therapy recommendation, n=121 and without therapy recommendation, n=148).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>PRISMA flow diagram of the corpus construction process. MII: medical informatics initiative; PRISMA: Preferred Reporting Items for Systematic Reviews and Meta-Analyses; SHIP: smart hospital information platform.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e99136_fig02.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Descriptive corpus statistics for the parallel and MII<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> corpora.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Documents, n (%)</td><td align="left" valign="bottom">Sections, n (%)</td><td align="left" valign="bottom" colspan="3">Words per documents</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Mean (SD)</td><td align="left" valign="top">Median (IQR)</td><td align="left" valign="top">Minimum-Maximum</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">Parallel</td></tr><tr><td align="left" valign="top">&#x2003;Protocols</td><td align="left" valign="top">47 (100)</td><td align="left" valign="top">N/A<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">881.8 (264.1)</td><td align="left" valign="top">849.0 (694.0-1097.0)</td><td align="left" valign="top">397&#x2010;1493</td></tr><tr><td align="left" valign="top">&#x2003;&#x2003;w/ TR<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">22 (46.8)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">1010.0 (270.4)</td><td align="left" valign="top">996.0 (809.3-1246.0)</td><td align="left" valign="top">582&#x2010;1493</td></tr><tr><td align="left" valign="top">&#x2003;&#x2003;w/o TR<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">25 (53.2)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">768.9 (203.3)</td><td align="left" valign="top">735.0 (642.0-916.0)</td><td align="left" valign="top">397&#x2010;1112</td></tr><tr><td align="left" valign="top">&#x2003;Patient protocols</td><td align="left" valign="top">47 (100)</td><td align="left" valign="top">410 (100)</td><td align="left" valign="top">558.0 (264.7)</td><td align="left" valign="top">464.0 (312.0-736.5)</td><td align="left" valign="top">229&#x2010;1221</td></tr><tr><td align="left" valign="top">&#x2003;&#x2003;w/ TR</td><td align="left" valign="top">22 (46.8)</td><td align="left" valign="top">227 (55.4)</td><td align="left" valign="top">780.9 (185.4)</td><td align="left" valign="top">736.5 (681.0-877.3)</td><td align="left" valign="top">464&#x2010;1221</td></tr><tr><td align="left" valign="top">&#x2003;&#x2003;w/o TR</td><td align="left" valign="top">25 (53.2)</td><td align="left" valign="top">183 (44.6)</td><td align="left" valign="top">361.9 (136.2)</td><td align="left" valign="top">319.0 (273.0-432.0)</td><td align="left" valign="top">229&#x2010;788</td></tr><tr><td align="left" valign="top" colspan="6">MII</td></tr><tr><td align="left" valign="top">&#x2003;Protocols</td><td align="left" valign="top">269 (100)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">1055.6 (340.4)</td><td align="left" valign="top">1021.0 (777.0-1277.0)</td><td align="left" valign="top">447&#x2010;2130</td></tr><tr><td align="left" valign="top">&#x2003;&#x2003;w/ TR</td><td align="left" valign="top">121 (45.0)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">1077.5 (331.1)</td><td align="left" valign="top">1065.0 (831.0-1301.0)</td><td align="left" valign="top">468&#x2010;2113</td></tr><tr><td align="left" valign="top">&#x2003;&#x2003;w/o TR</td><td align="left" valign="top">148 (55.0)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">1037.7 (347.9)</td><td align="left" valign="top">947.0 (773.3-1248.8)</td><td align="left" valign="top">447&#x2010;2130</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>MII: medical informatics initiative.</p></fn><fn id="table1fn2"><p><sup>b</sup>N/A: not applicable.</p></fn><fn id="table1fn3"><p><sup>c</sup>w/ TR: protocols with therapy recommendation. </p></fn><fn id="table1fn4"><p><sup>d</sup>w/o TR: protocols without therapy recommendation.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-2"><title>Patient Protocol Generation</title><sec id="s2-2-1"><title>LLMs</title><p>As an initial step, multiple open-weight instruction-following LLMs were evaluated, served via vLLM [<xref ref-type="bibr" rid="ref14">14</xref>] and queried using the OpenAI <italic>Python</italic> package [<xref ref-type="bibr" rid="ref15">15</xref>], with inference parameters configured per official documentation (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The models were selected based on 3 criteria: strong general-purpose generation capabilities, competitive multilingual coverage including German, and the feasibility of deployment on a single NVIDIA A100 (80 GB). Models exceeding 70B parameters were executed in 4-bit AWQ (activation-aware weight quantization) [<xref ref-type="bibr" rid="ref16">16</xref>], except for the OpenAI models, which were natively released in MXFP4 (microscaling floating point 4) quantization [<xref ref-type="bibr" rid="ref17">17</xref>]. The evaluated models deliberately span high-capacity state-of-the-art baselines and efficient small- to medium-sized models to quantify quality-efficiency trade-offs under realistic hardware constraints. These are summarized in <xref ref-type="table" rid="table2">Table 2</xref>.</p><p>This study was designed and reported in accordance with the TRIPOD-LLM (Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis) [<xref ref-type="bibr" rid="ref18">18</xref>] guideline for studies evaluating LLMs in health care.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Overview of the open-weight instruction-tuned LLMs<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> used for evaluation.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Creator</td><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Short name</td><td align="left" valign="bottom">Parameter count</td><td align="left" valign="bottom">Reasoning</td></tr></thead><tbody><tr><td align="left" valign="top">OpenAI</td><td align="left" valign="top">gpt-oss-120b [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">G-120B</td><td align="left" valign="top">120B</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">OpenAI</td><td align="left" valign="top">gpt-oss-20b [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">G-20B</td><td align="left" valign="top">20B</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Meta</td><td align="left" valign="top">Llama-3.3-70B-Instruct [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">L-70B</td><td align="left" valign="top">70B</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Meta</td><td align="left" valign="top">Llama-3.1-8B-Instruct [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">L-8B</td><td align="left" valign="top">8B</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Mistral AI</td><td align="left" valign="top">Mistral-Large-Instruct-2411 [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">M-123B</td><td align="left" valign="top">123B</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Mistral AI</td><td align="left" valign="top">Magistral-Small-2509 [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">M-24B</td><td align="left" valign="top">24B</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Alibaba Cloud</td><td align="left" valign="top">Qwen3-Next-80B-A3B-Instruct [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">Q-80B</td><td align="left" valign="top">80B</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Alibaba Cloud</td><td align="left" valign="top">Qwen3-4B-Thinking-2507 [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">Q-4B</td><td align="left" valign="top">4B</td><td align="left" valign="top">Yes</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-2-2"><title>Structured Generation</title><p>MTB patient protocols follow defined section schemes, and structural malformation errors have been shown to have a significantly negative impact on clinical evaluation scores [<xref ref-type="bibr" rid="ref24">24</xref>]. Two straightforward approaches for enforcing schema-compliant protocol generation were considered and rejected. Section-wise prompting, as used in our preliminary work, generates each section independently and cannot guarantee narrative coherence across the full protocol. Naive prompt-based template filling, where a protocol template is provided in the prompt and the model is instructed to complete it, proved brittle in pilot runs, as it produced altered headers, superfluous sections, and omitted mandatory fields.</p><p>Structured generation addresses both limitations by enforcing schema compliance at the token level through constrained decoding. Rather than instructing the model via natural language, constrained decoding, implemented through llguidance [<xref ref-type="bibr" rid="ref25">25</xref>], mathematically prevents the model from generating any output that violates a predefined format. Separate JSON schemas were defined for protocols with and without therapy recommendation. As fixed structural elements are guaranteed by the schema, clinician review is limited to dynamic content sections only. This minimizes the oversight burden. The final protocols were rendered from the structured model output using a Jinja2 [<xref ref-type="bibr" rid="ref26">26</xref>] template.</p></sec><sec id="s2-2-3"><title>Experiments: Comparison of Zero-Shot and One-Shot Approach</title><p>Comparison of zero-shot and one-shot approaches was a key component of the analysis. As the available data precluded fine-tuning, both approaches used prompt-based control, which mirrors the constraints common in hospital settings. Complete prompt templates and schemas are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><p>Regarding approach 1, structured template filling (A1) used zero-shot, schema-driven generation without examples. A system prompt specified persona, rules, and style constraints. The user prompt contained the generation instruction, short field-level instructions injected via the JSON schema, and the MTB protocol as source content.</p><p>Regarding approach 2, style-conditioned structured template filling (A2) extended A1 with one-shot, style-conditioned generation, motivated by in-context learning [<xref ref-type="bibr" rid="ref6">6</xref>]. The same schema and structured decoding were used, with one expert-written example injected per dynamic field, following the field instructions to anchor tone, lexical choices, and level of detail to institutional practices.</p></sec></sec><sec id="s2-3"><title>Evaluation</title><sec id="s2-3-1"><title>Automatic Evaluation</title><p>Given the limited gold standards and the requirement for German-suitable metrics, automatic evaluation combined reference-based fidelity metrics with reference-free readability and complexity measures. Reference-based metrics were computed for the 410 protocol sections by aligning the generated JSON fields with the corresponding sections in the expert-written protocols. These reference protocols were written using a structured framework grounded in didactic and psychological principles, codeveloped by MTB physicians, psychologists, and the WTZ patient advisory board [<xref ref-type="bibr" rid="ref13">13</xref>]. Reference-based metrics should therefore be interpreted as measures of stylistic and structural proximity to this reference, not as absolute indicators of lay-language quality.</p><p>Lexical fidelity was assessed using ROUGE-1 (Recall-Oriented Understudy for Gisting Evaluation) [<xref ref-type="bibr" rid="ref27">27</xref>], which measures the word-level overlap between generated and reference texts (0&#x2010;1, higher is better).</p><p>Semantic fidelity was assessed using BERTScore-F1 (Bidirectional Encoder Representations From Transformers Score) [<xref ref-type="bibr" rid="ref28">28</xref>], which computes embedding-level similarity to capture paraphrases and conceptual overlap (0&#x2010;1, higher is better).</p><p>Readability metrics can be applied without gold standards because they rely solely on text features, such as sentence and word length. Readability was assessed using Wiener Sachtextformel version 4 (WSTF4) [<xref ref-type="bibr" rid="ref29">29</xref>], a formula optimized for German, computed from the MS (percentage of words with 3 or more syllables) and the SL (average sentence length in words):</p><disp-formula id="E1"><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>W</mml:mi><mml:mi>S</mml:mi><mml:mi>T</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0.2744</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mo>+</mml:mo><mml:mn>0.2656</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mi>S</mml:mi><mml:mi>L</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1.1693</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Scores of 4&#x2010;6 indicate very easy texts, 7&#x2010;10 well-understood, 10&#x2010;12 medium difficulty, and above 12 specialist-level; scores can additionally be interpreted as the approximate school grade level required for comprehension.</p><p>Text complexity was assessed using a DistilBERT (Distilled Version of Bidirectional Encoder Representations From Transformers) model fine-tuned on German texts [<xref ref-type="bibr" rid="ref30">30</xref>]. The model yields scores on a 1&#x2010;7 scale, with lower values indicating simpler texts.</p><p>The differences in verbosity were quantified as the relative length differences between the generated and gold standard sections in words. A verbosity score of 1&#x2212;|&#x0394;length| was used, where higher values indicate lower deviation. This score is a proximity measure with a range of (&#x2212;&#x221E;, 1], and no clipping or normalization was applied. All verbosity results are reported as medians, which remain robust against the extreme values caused by overgeneration.</p></sec><sec id="s2-3-2"><title>Error Taxonomy</title><p>Automated metrics provide a scalable proxy for linguistic fluency but are notoriously brittle in clinical contexts due to their inability to capture medical safety implications or the specific error types inherent in LLM-generated outputs.</p><p>Rigorous evaluation of patient protocols requires a taxonomy that covers factual correctness, communicative adequacy, and patient safety. Established frameworks such as the Physician Documentation Quality Instrument-9 [<xref ref-type="bibr" rid="ref31">31</xref>] or SummEval [<xref ref-type="bibr" rid="ref32">32</xref>] have been proven useful, yet both assign scores to quality criteria rather than pinpointing specific errors, a level of detail that more closely reflects the correction work clinicians actually perform. Error annotation frameworks proposed for summarization [<xref ref-type="bibr" rid="ref33">33</xref>], translation [<xref ref-type="bibr" rid="ref34">34</xref>], and clinical documentation [<xref ref-type="bibr" rid="ref10">10</xref>] offer finer-grained analysis but treat any addition beyond the source document as an error or hallucination. This assumption represents a fundamental mismatch for patient protocol generation, where explanatory additions are necessary to achieve health literacy alignment. Penalizing such elaborations would systematically underestimate quality and mischaracterize safe model behavior.</p><p>Therefore, a specific taxonomy was developed. The taxonomy proposed here classifies each identified issue along 3 dimensions: 4 error types (factual, omission, noise, and language), 2 contextual scopes (internal and external), and 3 severity levels (minor, major, and critical). Reports that contain at least one critical error are rated as invalid and not acceptable for clinical use.</p><p>The scope dimension reflects the dual information sources of patient protocols: the factual content of the MTB protocol (internal) and the general medical background knowledge required to make the content comprehensible to lay readers (external). This distinction prevents the misclassification of clinically necessary explanatory additions as hallucinations or factual errors.</p><p>The four error types are defined as follows: (1) factual errors occur when the text contradicts either the source document (internal), such as incorrect diagnoses or therapy recommendations, or established medical knowledge (external), such as a misdefined biomarker; (2) omission errors occur when relevant information is absent. Internal omissions cover patient-specific details from the source document, and external omissions cover missing background explanations necessary for lay understanding; (3) noise errors occur when a text contains unnecessary or misleading content. Internal noise comprises redundant details from the source document, and external noise refers to unrelated medical information that dilutes key messages; and (4) language errors are linguistic or stylistic issues (eg, grammar, syntax, or phrasing) that do not affect factual content but may reduce readability, professionalism, or patient trust.</p><p>Severity reflects the potential impact on patient understanding and safety. Minor errors involve terminological or stylistic inaccuracies with no safety implications. Major errors substantially impair lay comprehension or distort key medical facts. The patient may misunderstand content, yet no direct harm or dangerous consequence is likely to result. Critical errors carry direct patient safety implications and risk harm through misinformation, such as incorrect therapy instructions or the omission of essential safety warnings. Annotators must assign the highest applicable severity level.</p></sec><sec id="s2-3-3"><title>Expert Evaluation</title><p>Following best practices from machine translation human evaluation research, only domain experts participated in the manual evaluation, given the interpretive complexity of clinical content. Seven board-certified medical oncologists or hematologists from the WTZ served as annotators, with direct experience in complex MTB cases constituting relevant domain expertise.</p><p>A stratified random sample of 25 MTB cases was drawn from the parallel corpus (20 with therapy recommendations and 5 without therapy recommendations). Priority was given to cases with therapy recommendations to ensure a focus on high-complexity reports with more dynamic fields and greater error risk. Consequently, expert-level characterization of protocols without therapy recommendations is more limited. To avoid distortion, there was no &#x201C;human-in-the-loop&#x201D; before the expert evaluation. Using the top-performing model from the automatic evaluation, paired A1 and A2 outputs yielded 50 unique protocols. Restricting expert evaluation to a single model was a deliberate choice, as including multiple models was not feasible given the annotation workload, and the primary goal was to compare generation approaches rather than models. Each protocol was independently rated by 2 clinicians, for a total of 100 ratings. As no consensus discussion or Delphi reconciliation was performed, interannotator agreement was quantified post hoc using Gwet agreement coefficient 1 (AC1), as described in the Statistical Analysis section.</p><p>A between-subject blinded design preserved ecological validity: each annotator saw each MTB protocol only once, and the generation conditions (A1 vs A2) were undisclosed. This mirrors anticipated routine deployment conditions, where each clinician reviews a given case only once and thereby avoids potential fatigue effects.</p><p>Clinical usability was assessed along the 3 ISO 9241&#x2010;11 usability dimensions of effectiveness, efficiency, and satisfaction using 3 complementary instruments.</p><list list-type="order"><list-item><p>Paragraph-level error annotation (effectiveness): annotators applied the error taxonomy described above at the paragraph level. They annotated error type, scope, and severity. This granular approach captures both specific error profiles and their potential clinical impact.</p></list-item><list-item><p>Perceived postediting effort (PPEE and efficiency): drawing on effort expectancy theory [<xref ref-type="bibr" rid="ref35">35</xref>] and machine translation postediting research [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>], clinicians rated the anticipated workload to bring each protocol to patient-ready quality on a 5-point Likert scale (1=very low, 5=very high): &#x201C;How much effort would be required to edit this protocol to make it suitable as a patient protocol?&#x201D;</p></list-item><list-item><p>Net promoter score [<xref ref-type="bibr" rid="ref38">38</xref>] (NPS and satisfaction): Clinicians rated their overall satisfaction using the NPS on a 0&#x2010;10 scale; they answered a single item: &#x201C;How likely are you to recommend the AI-generated draft shown above to a colleague for creating a patient protocol?&#x201D;</p></list-item></list><p>Together, these 3 instruments capture distinct and decision-relevant aspects of usability: error annotation reflects objective output quality, PPEE proxies operational integration burden, and NPS captures broader clinical acceptability.</p><p>A dedicated web interface built using Streamlit [<xref ref-type="bibr" rid="ref39">39</xref>] supported expert evaluation. Hosted within the hospital network and accessible via institutional credentials, it comprised 3 components: a project overview describing this study and error taxonomy, the core evaluation interface, and a visual decision tree guiding annotators through taxonomy category assignment. The evaluation interface used a dual-pane layout. The left panel displayed the original MTB protocol; the right panel presented the generated patient protocol divided into dynamic sections. Annotators first read the full protocol and then assessed each dynamic section for errors. Multiple errors could be recorded per section with optional free-text comments. After completing section-level annotation, annotators rated the PPEE and NPS for the full document. All annotations were saved as JSON files for subsequent analysis.</p><p>Annotators took part in a training procedure that included an online session, 2 written reference documents, and 5 practice protocols not part of the final evaluation set. During the online session, this study&#x2019;s team walked through the taxonomy and the interface and answered annotators&#x2019; questions directly. One reference document explained the error taxonomy with definitions and worked examples. The other described the evaluation interface. A decision tree for step-by-step category assignment was embedded in both the interface and the taxonomy document and was available throughout annotation.</p></sec></sec><sec id="s2-4"><title>Statistical Analysis</title><sec id="s2-4-1"><title>Automatic Evaluation</title><p>For each system (defined as LLM&#x00D7;prompting approach) and each metric, the results are summarized as the median with a 95% percentile bootstrap CI (5000 resamples, stratified by document).</p><p>System-level comparisons used pairwise tournament analysis to evaluate all system combinations head-to-head across metrics. For each pair, the proportion of instances in which system i outperformed system j was computed. The net win percentages (wins minus losses) yield an interpretable scale-independent performance ranking.</p><p>The primary inferential analysis compared the 2 prompting approaches (A2 vs A1) within each LLM using the Wilcoxon signed-rank test [<xref ref-type="bibr" rid="ref40">40</xref>] on document-level paired differences (A2&#x2212;A1). The effect size was quantified using the matched-pairs rank-biserial correlation r, where <italic>r</italic>=1 indicates perfect superiority of A2 and r=&#x2212;1 indicates perfect inferiority. Benjamini-Hochberg correction [<xref ref-type="bibr" rid="ref41">41</xref>] controlled the false discovery rate. All raw <italic>P</italic> values (8 LLMs&#x00D7;5 metrics) were pooled into one correction family per analysis domain. Adjusted <italic>P</italic> values below .05 are considered statistically significant.</p><p>All analyses were conducted across 3 prespecified subsets (overall, with therapy recommendation, and without therapy recommendation) to account for potential heterogeneity in protocol structure and vocabulary between document types.</p><p>Analyses were implemented in Python (version 3.12.11), using NumPy, pandas, SciPy, and statsmodels.</p></sec><sec id="s2-4-2"><title>Expert Evaluation</title><p>Five effectiveness outcomes were captured at the document level: error count, severity-weighted error score (minor=1, major=2, and critical=3), and errors stratified by type, scope, and severity. Spearman rank-order correlation (&#x03C1;) quantified the association between the severity-weighted error score and expert ratings (NPS and PPEE). NPS was additionally analyzed using the standard 3-category classification (detractors: 0&#x2010;6, passives: 7&#x2010;8, and promoters: 9&#x2010;10).</p><p>For each approach and outcome, the results are reported as the median with a 95% percentile bootstrap CI (5000 resamples). Before the A1 vs A2 comparisons, error counts were averaged across the 2 raters to obtain a single case-level estimate.</p><p>The highly zero-inflated distribution of paired differences (eg, 17/25, 70%, identical outputs in the noise category) made the Wilcoxon signed-rank test inappropriate. Instead, an exact binomial sign test [<xref ref-type="bibr" rid="ref42">42</xref>] on nontied pairs assessed pairwise approach preference, with Cohen g reported as the effect size measure. Benjamini-Hochberg correction controlled the false discovery rate across all outcomes within each analysis domain. Adjusted <italic>P</italic> values below .05 are considered statistically significant.</p><p>Interannotator agreement on error presence was evaluated using Gwet AC1 [<xref ref-type="bibr" rid="ref43">43</xref>]. This metric was deliberately chosen to counter the highly imbalanced nature of the evaluation data, where the absence of errors constituted the overwhelming majority of the observations. In such skewed distributions, metrics such as Krippendorff &#x03B1; suffer from the prevalence paradox, severely underestimating reliability. All coefficients included bootstrapped 95% CIs (5000 resamples).</p><p>Analyses were implemented in Python (version 3.12.11), using NumPy, SciPy, pandas, and the irrCAC library.</p></sec></sec><sec id="s2-5"><title>Ethical Considerations</title><p>The institutional ethics committee of the Medical Faculty, University of Duisburg-Essen, approved this study, with no ethical or legal concerns raised (reference: 24&#x2010;12093-BO; June 16, 2025). Data use was restricted to MTB protocols from patients who consented within the Medical Informatics Initiative broad consent; no patient recontact or intervention occurred. All data processing, model inference, and evaluation tools were hosted on secure institutional servers within the University Hospital network to ensure regulatory compliance.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Automatic Evaluation</title><p>In the pairwise tournament (<xref ref-type="fig" rid="figure3">Figure 3</xref>), Llama-3.3-70B-Instruct (L-70B) achieved the highest net win percentage across all metrics in both approaches (+14.9 pp under A1 and +9.5 pp under A2), while gpt-oss-20b (G-20B) ranked the lowest (&#x2212;11.5 pp under A1 and &#x2212;12.7 pp under A2). In both with- and without-therapy-recommendation subsets, the pairwise tournament rankings were broadly consistent; however, in cases without therapy recommendation, Mistral-Large-Instruct-2411 (M-123B) marginally outranked L-70B in the combined (A1+A2) tournament, whereas L-70B ranked first in cases with therapy recommendation.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Tournament ranking across datasets for A1 (A, left) and A2 (B, right). Each bar represents the mean net win percentage of a given LLM against all other LLMs, computed relative to an overall tie baseline of 50% in a pairwise metric comparison. Error bars indicate 95% bootstrap CIs. LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e99136_fig03.png"/></fig><p>The median metric values (<xref ref-type="table" rid="table3">Table 3</xref>) corroborated these trends. Under A1, the highest-scoring system was L-70B, and the lowest-scoring was G-20B. Under A2, L-70B again achieved the highest fidelity scores (ROUGE-1: 0.67 and BERTScore-F1: 0.63). A2 yielded significantly higher ROUGE-1 and BERTScore-F1 scores than A1 across all 8 models (<xref ref-type="table" rid="table4">Table 4</xref>). Verbosity scores show that generated protocols deviated 21%&#x2010;33% from the reference length in A1; A2 reduced this to 18%&#x2010;28%. Verbosity improvements from A1 to A2 were statistically significant for 5 of the 8 models, but not for L-70B, Llama-3.1-8B-Instruct (L-8B), or gpt-oss-120b (G-120B). Across all sections, models, and approaches, the verbosity score had a median of 0.8 (IQR 0.4&#x2010;1.0). The minimum was &#x2212;667, caused by a small number of pathologically long generations. As verbosity is reported as a median, these tail values did not affect the per-system scores.</p><p>The WSTF4 scores significantly improved from A1 to A2 for all models. The largest effects were observed for the models with the worst A1 scores: L-8B (<italic>r</italic>=0.53), G-120B (<italic>r</italic>=0.43), and G-20B (<italic>r</italic>=0.35). Under A2, L-70B, M-123B, and Magistral-Small-2509 (M-24B) fell marginally below the gold standard (10.73 and 10.72 vs 10.89). DistilBERT-based complexity assessments showed that all generated texts remained far from the gold standard (2.79) within a similar complexity range (4.10&#x2010;4.78), with no systematic differences between the approaches.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Median automatic evaluation metrics across datasets for approaches 1 and 2. The gold standard row reports the reference free metrics for expert-written patient protocols. Medians were computed for dynamic sections only.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">ROUGE-1 &#x2191;</td><td align="left" valign="bottom">BERTS-F1<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> &#x2191;</td><td align="left" valign="bottom">WSTF4<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> &#x2193;</td><td align="left" valign="bottom">Complexity<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup> &#x2193;</td><td align="left" valign="bottom">Verbosity &#x2191;</td></tr></thead><tbody><tr><td align="left" valign="top">Gold standard</td><td align="left" valign="top">N/A<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">N/A</td><td align="left" valign="top">10.89</td><td align="left" valign="top">2.79</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top" colspan="6">Approach 1, median (IQR)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>G-120B</td><td align="left" valign="top">0.50 (0.26-1.00)</td><td align="left" valign="top">0.45 (0.20-1.00)</td><td align="left" valign="top">11.97 (8.25-13.90)</td><td align="left" valign="top">4.72 (1.93-5.21)</td><td align="left" valign="top">0.79 (0.38-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>G-20B</td><td align="left" valign="top">0.40 (0.24-1.00)</td><td align="left" valign="top">0.39 (0.20-1.00)</td><td align="left" valign="top">11.93 (8.25-13.89)</td><td align="left" valign="top">4.36 (1.93-5.17)</td><td align="left" valign="top">0.70 (0.32-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L-70B</td><td align="left" valign="top">0.53 (0.25-1.00)</td><td align="left" valign="top">0.53 (0.22-1.00)</td><td align="left" valign="top">10.88 (8.25-13.12)</td><td align="left" valign="top">4.25 (1.93-4.98)</td><td align="left" valign="top">0.79 (0.36-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L-8B</td><td align="left" valign="top">0.40 (0.23-1.00)</td><td align="left" valign="top">0.48 (0.20-1.00)</td><td align="left" valign="top">11.67 (8.25-13.50)</td><td align="left" valign="top">4.10 (1.93-4.88)</td><td align="left" valign="top">0.67 (0.36-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>M-123B</td><td align="left" valign="top">0.40 (0.25-1.00)</td><td align="left" valign="top">0.44 (0.21-1.00)</td><td align="left" valign="top">10.94 (8.25-13.04)</td><td align="left" valign="top">4.18 (1.87-5.15)</td><td align="left" valign="top">0.67 (0.37-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>M-24B</td><td align="left" valign="top">0.42 (0.27-1.00)</td><td align="left" valign="top">0.40 (0.25-1.00)</td><td align="left" valign="top">10.92 (8.25-13.00)</td><td align="left" valign="top">4.34 (1.93-4.98)</td><td align="left" valign="top">0.68 (0.32-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q-80B</td><td align="left" valign="top">0.47 (0.29-1.00)</td><td align="left" valign="top">0.43 (0.23-1.00)</td><td align="left" valign="top">11.31 (8.25-13.09)</td><td align="left" valign="top">4.78 (1.93-5.24)</td><td align="left" valign="top">0.76 (0.21-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q-4B</td><td align="left" valign="top">0.40 (0.27-1.00)</td><td align="left" valign="top">0.43 (0.22-1.00)</td><td align="left" valign="top">11.68 (8.25-13.37)</td><td align="left" valign="top">4.39 (1.93-5.08)</td><td align="left" valign="top">0.78 (0.25-1.00)</td></tr><tr><td align="left" valign="top" colspan="6">Approach 2, median (IQR)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>G-120B</td><td align="left" valign="top">0.56 (0.32-1.00)</td><td align="left" valign="top">0.57 (0.27-1.00)</td><td align="left" valign="top">11.40 (8.25-13.15)</td><td align="left" valign="top">4.76 (1.91-5.19)</td><td align="left" valign="top">0.81 (0.42-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>G-20B</td><td align="left" valign="top">0.55 (0.29-1.00)</td><td align="left" valign="top">0.53 (0.25-1.00)</td><td align="left" valign="top">11.41 (8.25-13.22)</td><td align="left" valign="top">4.64 (1.91-5.19)</td><td align="left" valign="top">0.81 (0.39-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L-70B</td><td align="left" valign="top">0.67 (0.37-1.00)</td><td align="left" valign="top">0.63 (0.36-1.00)</td><td align="left" valign="top">10.73 (8.25-12.52)</td><td align="left" valign="top">4.46 (1.92-5.09)</td><td align="left" valign="top">0.80 (0.44-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L-8B</td><td align="left" valign="top">0.56 (0.34-1.00)</td><td align="left" valign="top">0.56 (0.32-1.00)</td><td align="left" valign="top">11.00 (8.25-12.77)</td><td align="left" valign="top">4.34 (1.93-5.10)</td><td align="left" valign="top">0.75 (0.42-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>M-123B</td><td align="left" valign="top">0.56 (0.33-1.00)</td><td align="left" valign="top">0.56 (0.33-1.00)</td><td align="left" valign="top">10.72 (8.25-12.82)</td><td align="left" valign="top">4.19 (1.89-5.13)</td><td align="left" valign="top">0.72 (0.38-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>M-24B</td><td align="left" valign="top">0.63 (0.39-1.00)</td><td align="left" valign="top">0.59 (0.36-1.00)</td><td align="left" valign="top">10.72 (8.25-12.50)</td><td align="left" valign="top">4.34 (1.93-5.16)</td><td align="left" valign="top">0.76 (0.43-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q-80B</td><td align="left" valign="top">0.57 (0.34-1.00)</td><td align="left" valign="top">0.56 (0.29-1.00)</td><td align="left" valign="top">10.92 (8.25-12.88)</td><td align="left" valign="top">4.71 (1.93-5.25)</td><td align="left" valign="top">0.82 (0.38-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q-4B</td><td align="left" valign="top">0.61 (0.37-1.00)</td><td align="left" valign="top">0.59 (0.32-1.00)</td><td align="left" valign="top">11.14 (8.25-13.04)</td><td align="left" valign="top">4.34 (1.93-5.11)</td><td align="left" valign="top">0.82 (0.49-1.00)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>BERTS-F1: Bidirectional Encoder Representations From Transformers Score.</p></fn><fn id="table3fn2"><p><sup>b</sup>WSTF4: Wiener Sachtextformel version 4.</p></fn><fn id="table3fn3"><p><sup>c</sup>Complexity: DistilBERT (Distilled Version of Bidirectional Encoder Representations From Transformers)-based complexity.</p></fn><fn id="table3fn4"><p><sup>d</sup>N/A: not applicable.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Median paired differences in evaluation metrics between approaches (approach 2&#x2212;approach 1) across datasets with effect sizes and <italic>P</italic> values.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">LLM<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="bottom" colspan="3">ROUGE-1</td><td align="left" valign="bottom" colspan="3">BERTS-F1<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="bottom" colspan="3">WSTF4<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="bottom" colspan="3">Complexity<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="bottom" colspan="3">Verbosity</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x0394;</td><td align="left" valign="top">r</td><td align="left" valign="top"><italic>P</italic></td><td align="left" valign="top">&#x0394;</td><td align="left" valign="top">r</td><td align="left" valign="top"><italic>P</italic></td><td align="left" valign="top">&#x0394;</td><td align="left" valign="top">r</td><td align="left" valign="top"><italic>P</italic></td><td align="left" valign="top">&#x0394;</td><td align="left" valign="top">r</td><td align="left" valign="top"><italic>P</italic></td><td align="left" valign="top">&#x0394;</td><td align="left" valign="top">r</td><td align="left" valign="top"><italic>P</italic></td></tr></thead><tbody><tr><td align="left" valign="top">G-120B</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.48</td><td align="left" valign="top">.02</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.66</td><td align="left" valign="top">.001</td><td align="left" valign="top">&#x2212;0.32</td><td align="left" valign="top">0.43</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.00</td><td align="left" valign="top">0.02</td><td align="left" valign="top">.73</td><td align="left" valign="top">0.04</td><td align="left" valign="top">0.31</td><td align="left" valign="top">.07</td></tr><tr><td align="left" valign="top">G-20B</td><td align="left" valign="top">0.03</td><td align="left" valign="top">0.63</td><td align="left" valign="top">.002</td><td align="left" valign="top">0.03</td><td align="left" valign="top">0.59</td><td align="left" valign="top">.003</td><td align="left" valign="top">&#x2212;0.25</td><td align="left" valign="top">0.35</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.10</td><td align="left" valign="top">&#x2212;0.34</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.10</td><td align="left" valign="top">0.57</td><td align="left" valign="top">.002</td></tr><tr><td align="left" valign="top">L-70B</td><td align="left" valign="top">0.03</td><td align="left" valign="top">0.77</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.02</td><td align="left" valign="top">0.72</td><td align="left" valign="top">.001</td><td align="left" valign="top">&#x2212;0.02</td><td align="left" valign="top">0.15</td><td align="left" valign="top">.03</td><td align="left" valign="top">0.19</td><td align="left" valign="top">&#x2212;0.53</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.00</td><td align="left" valign="top">&#x2212;0.09</td><td align="left" valign="top">.66</td></tr><tr><td align="left" valign="top">L-8B</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.59</td><td align="left" valign="top">.002</td><td align="left" valign="top">0.05</td><td align="left" valign="top">0.54</td><td align="left" valign="top">.005</td><td align="left" valign="top">&#x2212;0.50</td><td align="left" valign="top">0.53</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.21</td><td align="left" valign="top">&#x2212;0.54</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.05</td><td align="left" valign="top">0.32</td><td align="left" valign="top">.07</td></tr><tr><td align="left" valign="top">M-123B</td><td align="left" valign="top">0.09</td><td align="left" valign="top">0.70</td><td align="left" valign="top">.001</td><td align="left" valign="top">0.11</td><td align="left" valign="top">0.76</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x2212;0.18</td><td align="left" valign="top">0.21</td><td align="left" valign="top">.002</td><td align="left" valign="top">&#x2212;0.01</td><td align="left" valign="top">&#x2212;0.01</td><td align="left" valign="top">.90</td><td align="left" valign="top">0.04</td><td align="left" valign="top">0.50</td><td align="left" valign="top">.006</td></tr><tr><td align="left" valign="top">M-24B</td><td align="left" valign="top">0.05</td><td align="left" valign="top">0.84</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.84</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x2212;0.28</td><td align="left" valign="top">0.41</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.08</td><td align="left" valign="top">&#x2212;0.20</td><td align="left" valign="top">.003</td><td align="left" valign="top">0.09</td><td align="left" valign="top">0.63</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Q-80B</td><td align="left" valign="top">0.00</td><td align="left" valign="top">0.80</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.00</td><td align="left" valign="top">0.82</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x2212;0.22</td><td align="left" valign="top">0.33</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x2212;0.02</td><td align="left" valign="top">0.28</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.62</td><td align="left" valign="top">.002</td></tr><tr><td align="left" valign="top">Q-4B</td><td align="left" valign="top">0.13</td><td align="left" valign="top">0.88</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.12</td><td align="left" valign="top">0.86</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x2212;0.33</td><td align="left" valign="top">0.41</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x2212;0.10</td><td align="left" valign="top">0.20</td><td align="left" valign="top">.002</td><td align="left" valign="top">0.10</td><td align="left" valign="top">0.63</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table4fn2"><p><sup>b</sup>BERTS-F1: Bidirectional Encoder Representations From Transformers Score.</p></fn><fn id="table4fn3"><p><sup>c</sup>WSTF4: Wiener Sachtextformel version 4.</p></fn><fn id="table4fn4"><p><sup>d</sup>Complexity: DistilBERT (Distilled Version of Bidirectional Encoder Representations From Transformers)-based complexity.</p></fn></table-wrap-foot></table-wrap><p>In the fidelity-readability scatter plot (<xref ref-type="fig" rid="figure4">Figure 4</xref>), the A2 systems clustered toward higher BERTScore-F1 and lower WSTF4 values relative to their A1 counterparts. L-70B_A2 and M-123B_A2 showed the most favorable profiles, as they achieved a BERTScore-F1 &#x2265;.59 and a WSTF4 &#x2265;&#x2212;10.73.</p><p>A1-to-A2 improvements in lexical and semantic fidelity were more pronounced in cases with therapy recommendation. However, higher fidelity does not necessarily mean improved quality. It only reflects a closer alignment with the reference. Median paired differences in ROUGE-1 and BERTScore were significant across models in cases with therapy recommendation, whereas median differences were near zero and nonsignificant for all 8 models in cases without therapy recommendation. WSTF4 improvements were significant in both case types for all models, except L-70B for cases without therapy recommendation. Detailed results per case type can be found in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><p>Notably, neither the model parameter count nor the use of reasoning capabilities was associated with a consistent performance advantage across metrics.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Comparison of semantic similarity (BERTScore) and readability (WSTF4) across prompting approaches (A1: blue and A2: orange). WSTF4 is inverted, such that higher values indicate higher readability. Marker size indicates model size, and reasoning capabilities are marked with a diamond. BERTScore: Bidirectional Encoder Representations From Transformers Score; WSTF4: Wiener Sachtextformel version 4.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e99136_fig04.png"/></fig></sec><sec id="s3-2"><title>Clinical Usability</title><sec id="s3-2-1"><title>Effectiveness: Error Annotation</title><p>In 16% (230/1420) of annotated paragraphs, experts identified a total of 253 errors, which resulted in a median of 2.0 (IQR 1.5&#x2010;3.5) errors per protocol. By approach, A1 yielded a median of 2.0 (IQR 1.5&#x2010;2.5) errors per protocol and A2 a median of 2.5 (IQR 1.5&#x2010;3.5) errors per protocol. Median paired differences were <italic>&#x0394;</italic>=0.0 (95% CI &#x2212;0.5 to 2.0), with no significant differences observed in error counts between A1 and A2.</p><p>The proportion of protocols with &#x2265;1 critical error was 20% (n=5) for A1 and 40% (n=10) for A2. The median of critical errors per protocol was 0.0 (IQR 0.0&#x2010;0.5) for A1 and 0.5 (IQR 0.0&#x2010;1.5) for A2. Major errors occurred at a median of 0.5 (IQR 0.0&#x2010;0.5) per protocol for A1 vs 1.0 (IQR 0.0&#x2010;1.5) for A2, while minor errors occurred at 1.0 (IQR 0.5&#x2010;1.5) for both A1 and A2. Median paired differences in error severity between approaches were <italic>&#x0394;</italic>=0.0 (95% CI 0.0 to 1.0) for critical errors, <italic>&#x0394;</italic>=0.5 (95% CI 0.0 to 1.0) for major errors, and &#x0394;=&#x2212;0.5 (95% CI &#x2212;1.0 to 0.5) for minor errors.</p><p>Critical errors fell into 2 main categories. The largest group, with 24 annotations across 12 protocols, involved misjudgment of clinical relevance. This included providing explanations for therapeutically irrelevant alterations, omitting secondary options such as genetic counseling or trial referrals, and incorrectly characterizing the evidence base. These errors occurred under both prompting conditions, but they were more frequent under condition A2. The increase in critical errors under A2 was driven by exemplar contamination (18 annotations across 6 protocols). The model carried over content from the one-shot exemplar and manifested 3 subtypes: drug or therapy carryover (a drug featured in the exemplar was recommended for a patient for whom it was not indicated), approval-frame carryover (the exemplar&#x2019;s off-label framing caused the model to misrepresent the approval status of guideline-recommended therapies), and trial-name carryover (clinical trials cited in the exemplar were inserted in place of the patient&#x2019;s actual evidence base). <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref> contains a table listing all critical errors, their types, the sections of the protocols in which they occurred, and descriptions of the deidentified content.</p><p>While language was the prominent error type for A1 (40/108, 37%), factual errors were the most prominent type in A2 (69/145, 47.6%). The median of factual errors per protocol was 0.5 (IQR 0.0&#x2010;1.0) for A1 and 1.0 (IQR 0.5&#x2010;2.0) for A2. Omission errors occurred at a median of 0.5 (IQR 0.0&#x2010;1.0) per protocol for both A1 and A2, while noise errors occurred at 0.0 (IQR 0.0&#x2010;0.5) for A1 and 0.0 (IQR 0.0&#x2010;0.0) for A2. Language errors had a median of 0.5 (IQR 0.5&#x2010;1.0) per protocol for A1 and 0.5 (IQR 0.0&#x2010;1.0) for A2.</p><p>However, median paired differences showed no notable differences in error types between A1 and A2 (factual &#x2206;=0.0, 95% CI 0.0 to 1.5, omission &#x2206;=0.0, 95% CI 0.0 to 0.5, and noise &#x2206;=0.0, 95% CI 0.0 to 0.0) except for language with &#x2206;=&#x2212;0.5 (95% CI &#x2212;1.0 to 0.0) while not significant.</p><p><xref ref-type="fig" rid="figure5">Figure 5</xref> shows the distributions of the severity-weighted error scores per protocol. It shows that for factual errors the median paired difference after severity weighting changed to &#x2206;=0.5 (95% CI &#x2212;0.5 to 4.0). For the other types, there were no notable changes when severity weighting was applied, except for the minimally increased CIs for omission and language.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Comparison of severity-weighted error scores per protocol (minor=1, major=2, and critical=3) across prompting approaches (A1: blue and A2: orange), with annotated median (x&#x0303;), mean (diamond), median paired difference (&#x0394;x&#x0303;), its CI, and effect size of the exact binomial test. Significant differences are marked with asterisks.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e99136_fig05.png"/></fig><p>Looking at the scope of errors (<xref ref-type="fig" rid="figure6">Figure 6</xref>), internal errors dominated under both approaches, constituting 88% (60/68) under A1 and 63% (71/112) under A2. The median internal errors per protocol were 1.0 (IQR 0.5&#x2010;1.5) for A1 and 1.0 (IQR 0.5&#x2010;2.5) for A2. External errors were rare under A1 (12% of all errors; median 0.0, IQR 0.0&#x2010;0.0) but increased markedly under A2 (37% of all errors; median 0.5, IQR 0.0&#x2010;1.5).</p><p>Overall, no significant differences were detected by the exact binomial test in the error type, severity, scope, or severity-weighted error scores between A1 and A2.</p><p>The interannotator agreement analysis yielded a substantial aggregate agreement (Gwet AC1=0.72, 95% CI 0.67 to 0.76) for all errors regardless of type, severity, and scope. Almost perfect agreement [<xref ref-type="bibr" rid="ref44">44</xref>] was reached on critical errors (Gwet AC1=0.94, 95% CI 0.92 to 0.96), which supports the validity of the critical error rate as the safety-relevant outcome of this study. Detailed reliability statistics, stratified by error type, severity, and scope, are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Comparison of error per protocol by scope across prompting approaches (A1: blue and A2: orange), with annotated median (x&#x0303;), mean (diamond), median paired difference (&#x0394;x&#x0303;), its CI, and effect size of the exact binomial test. Significant differences are marked with asterisks.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e99136_fig06.png"/></fig></sec><sec id="s3-2-2"><title>Efficiency: PPEE</title><p>Over all ratings, the PPEE (<xref ref-type="fig" rid="figure7">Figure 7A,C</xref>) had a median of 2.0 (95% CI 2.0 to 3.0; IQR 2.0&#x2010;3.0). A total of 52% (52/100) of ratings indicated very low to low effort, 24% (24/100) medium, and 24% (24/100) high to very high. For A1, the median was 2.5 (95% CI 2.0 to 3.0; IQR 2.0&#x2010;3.0), and for A2, the median was 3.0 (95% CI 2.5 to 3.5; IQR 2.0&#x2010;3.5). The median paired difference was <italic>&#x0394;</italic>=0.5 (95% CI &#x2212;0.5 to 1.0) and not significant. There was a moderate and significant (<italic>P</italic>&#x003C;.001) correlation between PPEE and severity-weighted error score (&#x03C1;=0.55).</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Comparison of PPEE (A and C) and NPS (B and D) ratings across prompting approaches. Green indicates better scores, and pink indicates worse scores. Boxplots of all ratings with annotated median (x&#x0303;) and mean (diamond). NPS: net promoter score; PPEE: perceived postediting effort.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e99136_fig07.png"/></fig></sec><sec id="s3-2-3"><title>Satisfaction: NPS</title><p>Across all 100 individual ratings, the NPS (<xref ref-type="fig" rid="figure7">Figure 7B,D</xref>) had a median of 7.0 (95% CI 6.0 to 8.0; IQR 5&#x2010;9). The category distribution classified 29% (29/100) of ratings as promoters (scores 9&#x2010;10), 25% (25/100) as passives (scores 7&#x2010;8), and 46% (46/100) as detractors (scores 0&#x2010;6). By approach, A1 had a median of 7.0 (95% CI 6.5 to 8.5; IQR 6.0&#x2010;9.0) compared to a median of 6.5 for A2 (95% CI 4.0 to 7.0; IQR 3.5&#x2010;7.0). The detractor share rose from 40% (20/50) under A1 to 52% (26/50) under A2. Median paired differences were &#x0394;=&#x2212;1.0 (95% CI &#x2212;3.0 to 0.5) and nonsignificant. Protocols rated by detractors carried a higher severity-weighted error burden, consistent with the moderate and significant (<italic>P</italic>&#x003C;.001) correlation between NPS and severity-weighted error score (&#x03C1;=&#x2212;0.54). Per rater, the detractor share ranged from 14% (2/14) to 86% (12/14), and the median item score from 4.0 (IQR 1.25-5.0) to 10.0 (IQR 10.0-10.0). The detractor predominance was concentrated in 2 raters who together contributed 24 of the 46 detractor ratings. Per-annotator distributions are reported in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings and Comparison to Prior Work</title><p>Our study evaluated the clinical usability of structured LLM-generated lay-language patient protocols under real-world conditions. While style-conditioned prompting improved automatic metrics (A2), it simultaneously increased clinically relevant error rates compared with zero-shot generation (A1).</p><p>Therefore, the most consequential finding was the systematic dissociation between automatic metric performance and expert-assessed clinical quality. This extends the critique of ROUGE and BERTScore as poorly calibrated with respect to factual consistency in general-domain abstractive summarization [<xref ref-type="bibr" rid="ref45">45</xref>].</p><p>A2 produced statistically significant improvements in lexical fidelity, semantic alignment, and surface readability across all 8 models. The asymmetric improvement across case types (with vs without therapy recommendation) may be caused by protocols with therapy recommendation containing more dynamic fields and higher within-corpus variation. These conditions provide more surface area for style anchoring to produce measurable changes.</p><p>These findings are consistent with those of prior work, which indicate that in-context learning can improve model outputs [<xref ref-type="bibr" rid="ref6">6</xref>]. This could suggest a preference for A2. However, expert annotation did not support this conclusion. Under A2, the proportion of protocols containing at least one critical error doubled (10/25, 40%, vs 5/25, 20%, under A1), the dominant error type shifted from language errors to factual errors, and the directional trends in both PPEE and NPS favored A1.</p><p>Although no individual comparison reached statistical significance, the consistent directional pattern across all expert-assessed outcomes (error severity, error type, postediting effort, and NPS) points to a coherent effect of prompting strategy rather than random variation. Overall, the prevalence of 20% (5/25) or 40% (10/25) of critical errors precludes the readiness of any approach for clinical use.</p><p>The mechanism underlying this dissociation is interpretable in terms of how in-context learning operates. Few-shot demonstrations primarily constrain output format and token distribution rather than conveying abstract task understanding [<xref ref-type="bibr" rid="ref46">46</xref>]. This may lead models to generalize factual content from examples beyond their appropriate scope. In a patient-specific clinical setting, an example anchors the lexical register effectively, while simultaneously introducing factual associations that may not apply to the current protocol. In a concrete example drawn from the expert evaluation dataset, the source protocol recommended evaluation for a trial targeting their specific mutation. However, the model recommended off-label use of Olaparib, a drug not mentioned in that patient&#x2019;s protocol but in the one-shot example. The protocol&#x2019;s actual recommendation, trial enrollment, was omitted entirely. This drug was likely not derived from general medical knowledge, but rather was carried over from the example. Similar patterns were detected in 18 other critical errors.</p><p>Constrained decoding enforces output format but not factual content, so a schema-compliant protocol can still carry these errors. A 2-stage architecture that first extracts and verifies the clinical attributes from the source protocol [<xref ref-type="bibr" rid="ref47">47</xref>], then generates the lay explanation from them, might mitigate this failure mode and is a direction for future work.</p><p>Model selection findings are relevant to the practical question of on-premises deployment. Llama-3.3-70B-Instruct achieved the strongest aggregate automatic metric performance and was selected for expert evaluation on this basis. However, neither parameter count nor the inclusion of reasoning capabilities was a reliable predictor of performance across metrics or models. This is consistent with early evidence that instruction-following quality is more predictive of task-specific performance than parameter scale alone [<xref ref-type="bibr" rid="ref48">48</xref>]. This specifically matters in the European health care context where closed-source models cannot be routinely applied to real patient data under current data protection regulations without institutional data processing agreements.</p><p>Despite significant WSTF4 improvements under A2, scores across both approaches remained in the medium difficulty range (10-12), which corresponds to an approximate comprehension requirement of at least grade 10. DistilBERT-based complexity scores remained between 4.10 and 4.78 across all models and approaches, against a gold standard of 2.79. This gap highlights a clinically relevant distinction between surface readability and conceptual density. The models can simplify sentence structures without simplifying the underlying genomic concepts. This may produce protocols that read grammatically fluently yet remain incomprehensible to patients with limited health literacy. Therefore, the reviewing clinician is still responsible for bridging the conceptual gap at a level appropriate for the health literacy of each patient. The DistilBERT complexity model was fine-tuned on general German texts and may not be fully calibrated to the clinical register; therefore, the absolute gap to the gold standard should be interpreted with caution. Closing this gap will likely require domain-adapted fine-tuning, personalization to individual patient health literacy, or more granular simplification instructions that specify expected reading level and explanation depth per dynamic field.</p><p>The error type shift between approaches also has implications for the efficiency dimension of clinical usability. Linguistic errors can be identified through proofreading a single document, whereas factual errors require active cross-referencing and increased workload. Critically, undetected factual errors carry the potential for seriously harming the patient.</p><p>Expert clinicians nonetheless rated most generated drafts as requiring low to moderate PPEE. This supports the premise that structured, open-weight generation under zero-shot conditions can provide meaningful drafting support. The median NPS indicated cautiously positive satisfaction ratings overall. However, detractors outnumbered promoters, which indicates that clinician acceptance is not uniform and that a substantial proportion of clinicians remain hesitant to recommend the generated drafts to colleagues. Two factors may account for this variance. The first is the quality of the patient letter drafts, as indicated by the negative correlation between NPS and severity-weighted error scores. The second factor is inherent rater preference. The wide variation in NPS scores among raters suggests that clinicians have different thresholds for recommending AI-generated assistance to their peers, even when evaluating drafts of equivalent technical quality. Future work is required to disentangle these factors, which would necessitate a dedicated study with fuller annotator crossing and more ratings per protocol.</p><p>The clinician-supervised drafting workflow assumes that expert reviewers will detect every critical error before patient delivery, though automation bias [<xref ref-type="bibr" rid="ref49">49</xref>] may compromise this in practice. In clinical decision support contexts, reviewers tend to over-trust AI-generated outputs and apply less scrutiny. Reviewers also perform worse under real workflow pressure. The favorable perceived effort ratings in this study may reflect this dynamic. This study did not test reviewer vigilance under realistic workflow conditions, and future work should assess error detection rates under ecologically valid review conditions before clinical deployment.</p></sec><sec id="s4-2"><title>Limitations</title><p>The specificity of MTBs constrains the generalizability of the findings of this study. MTB protocols represent a terminologically and structurally atypical genre of clinical documentation, and error profiles, readability baselines, and patient communication norms will differ substantially from other forms of documentation. The error taxonomy and evaluation framework are designed to be applicable more broadly, yet require empirical validation in other clinical contexts.</p><p>The expert evaluation was conducted with a single model (L-70B), selected on the basis of automatic metric performance. However, since this model was chosen using metrics that this study identified as imperfect proxies for clinical safety, the selection introduces a partial circularity. Therefore, the findings of the expert evaluation are conditional on this selection. Accordingly, the observed increase in critical errors under A2 is confirmed only for the L-70B model, while it is only hypothesized for other model families and institutions. Different error types or responses to prompting through the use of other models cannot be ruled out. Future work should therefore replicate the evaluation across model families. Retrieval-grounded evaluation [<xref ref-type="bibr" rid="ref50">50</xref>] could also be explored as a complement to surface-fidelity metrics, with the potential to capture evidence-based correctness that token-level scores miss.</p><p>Regarding data representativeness, the gold-standard protocols were authored by a single expert oncologist at one institution. While the protocols were developed according to a structured framework with input from communication specialists, medical educators, and the patient advisory board, they reflect one institutional practice and one clinician&#x2019;s writing style. Reference-based metrics such as ROUGE-1 and BERTScore therefore measure proximity to this single reference. They do not measure general lay-language quality. Higher scores under A2 should be read as increased stylistic alignment with the reference author, not as evidence of improved lay language quality. The underlying protocol format was therefore previously evaluated in the MyCODE study [<xref ref-type="bibr" rid="ref13">13</xref>] and showed high patient acceptance. Nevertheless, this concentration of authorship may be a form of reference bias and should be considered when interpreting reference-based metrics.</p><p>The annotation sample of 25 cases and 50 protocols was adequate for the primary inferential analyses but had limited statistical power for subgroup comparisons, particularly for sparse error categories such as noise. The highly zero-inflated distributions required conservative sign testing in place of the Wilcoxon procedure, which may have reduced sensitivity to true directional differences between approaches. The skewed distribution of case types in expert evaluation (20 with therapy recommendation and 5 without therapy recommendation) was intentional but leaves the quality of protocols without therapy recommendation under expert scrutiny undercharacterized.</p><p>The moderate, significant correlation between severity-weighted error scores and expert ratings offers some initial evidence for the plausibility of the evaluation framework. However, expert ratings remain subjective. Future work should therefore complement them with more objective measures and assess efficiency and satisfaction through real-time usage studies rather than relying on text quality as a proxy.</p><p>Finally, all evaluation data reflect the clinician&#x2019;s perspective rather than that of the intended recipient. The gold standard protocols were positively received by lay readers [<xref ref-type="bibr" rid="ref13">13</xref>]. However, this does not extend to AI-generated output. Whether generated protocols are comprehensible and useful to patients varying in health literacy and emotional state remains an open empirical question, which the preregistered study underway (MyCODEx: DRKS00037795) is designed to address.</p></sec><sec id="s4-3"><title>Conclusions</title><p>This study evaluated structured generation of German MTB patient protocols using open-weight LLMs under on-premises deployment constraints. Zero-shot and style-conditioned one-shot prompting strategies were implemented, with performance measured via both automatic and expert clinical evaluation. Three main conclusions emerge.</p><p>First, automatic metrics and expert clinical judgment diverged systematically by approach for the expert-evaluated model. Style-conditioned one-shot prompting improved all automatic metrics across all 8 models. However, for Llama-3.3-70B-Instruct, it also doubled the rate of protocols containing at least one critical error, with the dominant error type shifting from language to factual errors. This finding suggests that reference-based metrics may be insufficient as standalone quality assurance instruments, and that expert-led evaluation is essential.</p><p>Second, neither the model parameter count nor reasoning capabilities predicted performance reliably. Under the on-premises hardware constraints required by European data protection regulations, smaller, well-calibrated models approximated the performance of substantially larger counterparts, with direct implications for sustainable and privacy-preserving deployment planning.</p><p>Third, the formative evaluation of clinical usability across all 3 ISO 9241&#x2010;11 dimensions yielded cautiously positive, yet heterogeneous results across 25 expert-evaluated cases. While overall error rates were low and efficiency ratings favorable, the near-equal split between promoters and detractors indicates that clinician acceptance is not uniform. A persistent semantic complexity gap between generated and expert-written protocols may further limit communicative adequacy for lay readers. A critical error rate of 20%&#x2010;40% precludes clinical applicability of the selected model under the conditions tested.</p><p>Taken together, structured, zero-shot LLM generation may provide a useful foundation for drafting patient-facing MTB protocols. However, these outputs require mandatory review by experts with domain expertise sufficient to verify clinical claims against the source protocol (internal) and general medical knowledge (external). The results of this study are interpretable only in an expert-supervised context, as the acceptable error threshold for clinician-reviewed drafts differs fundamentally from that required for autonomous transmission. Given the limited scale of the expert evaluation, these conclusions should be read as formative evidence informing future development rather than as a definitive clinical validation of deployment readiness.</p><p>Future work should focus on patient-centered outcome evaluation and real-world assessment of editing effort and clinical acceptance, where workflow integration will be critical for adoption. In addition, multisite validation across different clinical protocol types and the development of domain-adapted training strategies will be essential to support broader and more reliable implementation.</p></sec></sec></body><back><ack><p>We would like to thank Kilian Elfert for his expert statistical consultation and invaluable feedback on the development of the statistical analysis plan. We gratefully acknowledge the contribution of the members of the West German Cancer Center (WTZ) Patient Advisory Board, who provided valuable feedback on the patient protocols and were actively involved in the design, conduct, and interpretation of the formative study. The data for this project were provided by the smart hospital information platform (SHIP), managed by the Data Integration Center at the University Medicine Essen. SHIP serves as a comprehensive digital health platform for integrating data from all major clinical subsystems using a holistic Fast Healthcare Interoperability Resources&#x2013;based approach. This enables the purification, analysis, distribution, and visualization of clinical data. Disclosure of delegation to generative AI (GenAI): the authors declare the use of GenAI in the research and writing processes. According to the GAIDeT (2025; Generative Artificial Intelligence Delegation Taxonomy), the following tasks were delegated to GenAI tools under full human supervision: literature search and systematization, development of experimental or research protocols, code generation, code optimization, creation of algorithms for data analysis, visualization, text generation, proofreading and editing, and summarizing text. The GenAI tools used were ChatGPT 5.2 (OpenAI), DeepL Write, and GitHub Copilot (Microsoft Corp). Responsibility for this final paper lies entirely with the authors. GenAI tools are not listed as authors and do not bear responsibility for the outcomes. Declaration submitted by TMGP</p></ack><notes><sec><title>Funding</title><p>This work received funding from KITE (Plattform f&#x00FC;r KI [K&#x00FC;nstliche Intelligenz]-Translation Essen) from the REACT-EU initiative (EFRE-0801977) [<xref ref-type="bibr" rid="ref51">51</xref>]. The work of TMGP and NB was funded by a PhD grant from the Deutsche Forschungsgemeinschaft, German Research Foundation (DFG) Research Training Group 2535 Knowledge- and data-based personalization of medicine at the point of care (WisPerMed). This research was codeveloped with the Collaborative Research Centre (CRC) Treatment expectation, funded by the German Research Foundation (DFG, project-ID 422744262&#x2013;TRR 289). The funders had no role in the study design, data collection, analysis, interpretation, or the writing of this paper.</p></sec><sec><title>Data Availability</title><p>The data supporting the findings of this study are not publicly available due to patient privacy and data protection regulations. Requests for access to restricted data should be directed to the corresponding author and must include a brief description of their intended use. Data sharing will require a formal data transfer agreement and may be subject to local ethical and legal reviews. The annotation platform, including deployment and configuration instructions and analysis scripts used to generate the results and figures in this paper, will be made publicly available in a permanent archive (Zenodo); the DOI will be provided in the final version.</p></sec></notes><fn-group><fn fn-type="con"><p>TMGP, NB, AF, SB, NP, KK, and IP conceptualized this study. TMGP developed the methodology, with contributions from NB, AF, CMF, and IP. TMGP handled data curation and formal analysis, developed software, and created visualizations. The investigation and validation were performed by TMGP, GZ, TG, MW, MA, MP, VR, and TH. AF, SB, NP, GZ, TG, MW, MA, MP, VR, TH, KK, DS, and IP provided the resources. TMGP drafted the original manuscript, and all authors contributed to the review and editing. CMF, PAH, DS, and IP supervised this study. AF managed the project administration.</p></fn><fn fn-type="conflict"><p>SB received honoraria for lectures from AstraZeneca, Gr&#x00FC;nenthal AG, Gr&#x00FC;nenthal GmbH, Lilly, and Janssen. TH received honoraria from Ipsen and received travel and accommodation costs from Janssen-Cilag.</p><p>MP received honoraria for advisory or consultancy roles from Amgen, AstraZeneca, Bayer, BMS, Lilly, Merck Healthcare Germany, MSD, Roche, Sanofi Aventis, Servier, Janssen Pharmaceuticals, Onkowissen, Pierre-Fabre, and Novartis; received honoraria from Amgen, AstraZeneca, BMS, Lilly, Merck Healthcare Germany, MSD, Sanofi Aventis, Servier, Roche, Onkowissen, and artTempi; received research funding from BMS, Lilly, Roche, and Amgen; and received other financial support from Amgen, AstraZeneca, BMS, Lilly, Merck Healthcare Germany, Roche, Sanofi Aventis, Servier, Pierre-Fabre, and artTempi.</p><p>DS received honoraria for consultation from BMS, Novartis, MSD, Moderna, BioNTech, Regeneron, Pierre Fabre, Pfizer, Boehringer Ingelheim, Replimune, Sunpharma, Philogen, Skyline Dx, Daiichi Sankyo, Ipsen, IoVance, IOBioTech, Immunocore, BioAlta, Immatics, Seagen, and Merck-Serono; received honoraria for participation on data safety monitoring boards or advisory boards for BMS, Novartis, MSD, Moderna, BioNTech, Regeneron, Pierre Fabre, Replimune, Philogen, Skyline Dx, Daiichi Sankyo, IMCheck, Ipsen, IoVance, IOBioTech, Immunocore, BioAlta, Immatics; received honoraria for lectures or presentations from BMS, Novartis, MSD, Pierre Fabre, Regeneron, Sunpharma, and Regeneron; received travel support from Pierre Fabre; has an unpaid leadership or fiduciary role for Dermatologic Cooperative Group (DeCOG), EuMelaREg, NVKH, and Deutsche Hautkrebsstiftung/Hiege-Stiftung. His institution received research funding from BMS, Amgen, MSD, and Novartis.</p><p>MS has no conflicts of interest in relation to this work. He received honoraria as a consultant from Amgen, AstraZeneca, Bristol Myers Squibb, Gilead, GlaxoSmithKline, Johnson &#x0026; Johnson, MSD, Novartis, Regeneron, Roche, and Sanofi, and received honoraria for continuing medical education (CME) presentations from Amgen, Bristol Myers Squibb, GlaxoSmithKline, Johnson &#x0026; Johnson, Lilly, MSD, and Roche. His institution receives research funding from Bristol Myers Squibb and Johnson &#x0026; Johnson.</p><p>MW received honoraria or had an advisory role for Amgen, AstraZeneca, BeOne Medicines, Bristol-Myers Squibb, Daiichi Sankyo, GlaxoSmithKline, Janssen, Lilly, neoConnect GmbH, Novartis, Pfizer, Roche, Takeda; received travel costs from Amgen, Janssen, Daiichi Sankyo, Roche; and received research funding from Amgen, Bristol-Myers Squibb, and Takeda.</p><p>IP served on advisory boards of BeOne, Daiichi Sankyo, Taiho Oncology Europe, and received honoraria from Deutsche Bundesbank, neoConnect GmbH, Daiichi Sankyo, Roche Diagnostics, Springer Medizin.</p><p>All other authors declare no competing interests.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AC1</term><def><p>agreement coefficient 1</p></def></def-item><def-item><term id="abb2">AWQ</term><def><p>activation-aware weight quantization</p></def></def-item><def-item><term id="abb3">BERTScore</term><def><p>Bidirectional Encoder Representations From Transformers Score</p></def></def-item><def-item><term id="abb4">DistilBERT</term><def><p>Distilled Version of Bidirectional Encoder Representations From Transformers</p></def></def-item><def-item><term id="abb5">ISO</term><def><p>International Organization for Standardization</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">MS</term><def><p>percentage of words with 3 or more syllables</p></def></def-item><def-item><term id="abb8">MTB</term><def><p>Molecular Tumor Board</p></def></def-item><def-item><term id="abb9">MXFP4</term><def><p> microscaling floating point 4</p></def></def-item><def-item><term id="abb10">NPS</term><def><p>net promoter score</p></def></def-item><def-item><term id="abb11">PPEE</term><def><p>perceived postediting effort</p></def></def-item><def-item><term id="abb12">ROUGE</term><def><p>Recall-Oriented Understudy for Gisting Evaluation</p></def></def-item><def-item><term id="abb13">SL</term><def><p>average sentence length in words</p></def></def-item><def-item><term id="abb14">TRIPOD</term><def><p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis</p></def></def-item><def-item><term id="abb15">WSTF4</term><def><p>Wiener Sachtextformel version 4</p></def></def-item><def-item><term id="abb16">WTZ</term><def><p>West German Cancer Center</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scholl</surname><given-names>I</given-names> </name><name name-style="western"><surname>Zill</surname><given-names>JM</given-names> </name><name name-style="western"><surname>H&#x00E4;rter</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dirmaier</surname><given-names>J</given-names> </name></person-group><article-title>An integrative model of patient-centeredness - a systematic review and concept analysis</article-title><source>PLOS ONE</source><year>2014</year><volume>9</volume><issue>9</issue><fpage>e107828</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0107828</pub-id><pub-id pub-id-type="medline">25229640</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Street</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Makoul</surname><given-names>G</given-names> </name><name name-style="western"><surname>Arora</surname><given-names>NK</given-names> </name><name name-style="western"><surname>Epstein</surname><given-names>RM</given-names> </name></person-group><article-title>How does communication heal? Pathways linking clinician-patient communication to health outcomes</article-title><source>Patient Educ Couns</source><year>2009</year><month>03</month><volume>74</volume><issue>3</issue><fpage>295</fpage><lpage>301</lpage><pub-id pub-id-type="doi">10.1016/j.pec.2008.11.015</pub-id><pub-id pub-id-type="medline">19150199</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Illert</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Stenzinger</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bitzer</surname><given-names>M</given-names> </name><etal/></person-group><article-title>The German Network for Personalized Medicine to enhance patient care and translational research</article-title><source>Nat Med</source><year>2023</year><month>06</month><volume>29</volume><issue>6</issue><fpage>1298</fpage><lpage>1301</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02354-z</pub-id><pub-id pub-id-type="medline">37280276</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bedi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Orr-Ewing</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Testing and evaluation of health care applications of large language models: a systematic review</article-title><source>JAMA</source><year>2025</year><month>01</month><day>28</day><volume>333</volume><issue>4</issue><fpage>319</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1001/jama.2024.21700</pub-id><pub-id pub-id-type="medline">39405325</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hoffmann</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rueger</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Current applications and challenges in large language models for patient care: a systematic review</article-title><source>Commun Med (Lond)</source><year>2025</year><month>01</month><day>21</day><volume>5</volume><issue>1</issue><fpage>26</fpage><pub-id pub-id-type="doi">10.1038/s43856-024-00717-2</pub-id><pub-id pub-id-type="medline">39838160</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Mann</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ryder</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Language models are few-shot learners</article-title><access-date>2026-07-04</access-date><conf-name>Proceedings of the 34th international conference on neural information processing systems Vancouver</conf-name><conf-date>Dec 6-12, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf">https://proceedings.neurips.cc/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf</ext-link></comment></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Azizi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language models encode clinical knowledge</article-title><source>Nature</source><year>2023</year><month>08</month><day>3</day><volume>620</volume><issue>7972</issue><fpage>172</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>OpenAI</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Adler</surname><given-names>S</given-names> </name><name name-style="western"><surname>Agarwal</surname><given-names>S</given-names> </name><etal/></person-group><article-title>GPT-4 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 4, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.08774</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Willard</surname><given-names>BT</given-names> </name><name name-style="western"><surname>Louf</surname><given-names>R</given-names> </name></person-group><article-title>Efficient guided generation for large language models</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 19, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2307.09702</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Asgari</surname><given-names>E</given-names> </name><name name-style="western"><surname>Monta&#x00F1;a-Brown</surname><given-names>N</given-names> </name><name name-style="western"><surname>Dubois</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A framework to assess clinical safety and hallucination rates of LLMs for medical text summarisation</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>13</day><volume>8</volume><issue>1</issue><fpage>274</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01670-7</pub-id><pub-id pub-id-type="medline">40360677</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pakull</surname><given-names>T</given-names> </name><name name-style="western"><surname>Dada</surname><given-names>A</given-names> </name><name name-style="western"><surname>Damm</surname><given-names>H</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Demner-Fushman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>D</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>P</given-names> </name></person-group><article-title>Preliminary evaluation of an open-source LLM for lay translation of german clinical documents</article-title><conf-name>Proceedings of the Second Workshop on Patient-Oriented Language Processing (CL4Health)</conf-name><conf-date>May 3-4, 2025</conf-date><conf-loc>Albuquerque, NM</conf-loc><fpage>180</fpage><lpage>192</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.cl4health-1.15</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="web"><article-title>Ergonomics of human-system interaction &#x2014; part 11: usability: definitions and concepts</article-title><source>ISO</source><year>2018</year><access-date>2026-07-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.iso.org/standard/63500.html">https://www.iso.org/standard/63500.html</ext-link></comment></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pretzell</surname><given-names>I</given-names> </name><name name-style="western"><surname>Fleischhauer</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mavroeidi</surname><given-names>I</given-names> </name><etal/></person-group><article-title>MyCODE: a prospective evaluation of lay-language molecular tumor board protocols in precision oncology</article-title><source>Oncologist</source><year>2026</year><month>04</month><day>10</day><volume>31</volume><issue>5</issue><fpage>oyag119</fpage><pub-id pub-id-type="doi">10.1093/oncolo/oyag119</pub-id><pub-id pub-id-type="medline">41915064</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kwon</surname><given-names>W</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhuang</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Efficient memory management for large language model serving with pagedattention</article-title><conf-name>SOSP &#x2019;23</conf-name><conf-date>Oct 23-26, 2023</conf-date><conf-loc>Koblenz, Germany</conf-loc><fpage>611</fpage><lpage>626</lpage><pub-id pub-id-type="doi">10.1145/3600006.3613165</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><article-title>Openai/openai-python</article-title><source>OpenAI</source><access-date>2026-07-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/openai/openai-python">https://github.com/openai/openai-python</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>G</given-names> </name><name name-style="western"><surname>Han</surname><given-names>S</given-names> </name></person-group><article-title>AWQ: activation-aware weight quantization for on-device LLM compression and acceleration</article-title><source>GetMobile: Mobile Comp Commun</source><year>2025</year><month>01</month><day>20</day><volume>28</volume><issue>4</issue><fpage>12</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1145/3714983.3714987</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Rouhani</surname><given-names>BD</given-names> </name><name name-style="western"><surname>Garegrat</surname><given-names>N</given-names> </name><name name-style="western"><surname>Savell</surname><given-names>T</given-names> </name><etal/></person-group><article-title>OCP microscaling formats (MX) specification - version 1.0</article-title><year>2023</year><month>09</month><access-date>2026-07-04</access-date><publisher-name>Open Compute Project</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.opencompute.org/documents/ocp-microscaling-formats-mx-v1-0-spec-final-pdf">https://www.opencompute.org/documents/ocp-microscaling-formats-mx-v1-0-spec-final-pdf</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gallifant</surname><given-names>J</given-names> </name><name name-style="western"><surname>Afshar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ameen</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The TRIPOD-LLM reporting guideline for studies using large language models</article-title><source>Nat Med</source><year>2025</year><month>01</month><volume>31</volume><issue>1</issue><fpage>60</fpage><lpage>69</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03425-5</pub-id><pub-id pub-id-type="medline">39779929</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><collab>OpenAI</collab><name name-style="western"><surname>Agarwal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ahmad</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ai</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Gpt-oss-120b &#x0026; gpt-oss-20b model card</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 8, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2508.10925</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Grattafiori</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dubey</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jauhri</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pandey</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kadian</surname><given-names>A</given-names> </name><name name-style="western"><surname>Al-Dahle</surname><given-names>A</given-names> </name></person-group><article-title>The llama 3 herd of models</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 23, 2024</comment><pub-id pub-id-type="doi">10.48550/ARXIV.2407.21783</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="web"><article-title>Mistralai/mistral-large-instruct-2411</article-title><source>Hugging Face</source><access-date>2026-07-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/mistralai/Mistral-Large-Instruct-2411">https://huggingface.co/mistralai/Mistral-Large-Instruct-2411</ext-link></comment></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Mistral</surname><given-names>AI</given-names> </name><name name-style="western"><surname>Rastogi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>AQ</given-names> </name><etal/></person-group><article-title>Magistral</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 12, 2025</comment><pub-id pub-id-type="doi">10.48550/ARXIV.2506.10910</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Li</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Qwen3 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  May 14, 2025</comment><pub-id pub-id-type="doi">10.48550/ARXIV.2505.09388</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Seo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Choi</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Evaluation framework of large language models in medical documentation: development and usability study</article-title><source>J Med Internet Res</source><year>2024</year><month>11</month><day>20</day><volume>26</volume><fpage>e58329</fpage><pub-id pub-id-type="doi">10.2196/58329</pub-id><pub-id pub-id-type="medline">39566044</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="web"><article-title>Guidance-ai/llguidance</article-title><source>GitHub</source><access-date>2026-07-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/guidance-ai/llguidance">https://github.com/guidance-ai/llguidance</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="web"><article-title>Pallets/jinja</article-title><source>GitHub</source><access-date>2026-07-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/pallets/jinja">https://github.com/pallets/jinja</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>CY</given-names> </name></person-group><article-title>ROUGE: a package for automatic evaluation of summaries</article-title><access-date>2026-07-04</access-date><conf-name>Workshop on Text Summarization Branches Out, Post-Conference Workshop of ACL</conf-name><conf-date>Jul 25-26, 2004</conf-date><conf-loc>Barcelona, Spain</conf-loc><fpage>74</fpage><lpage>81</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.microsoft.com/en-us/research/publication/rouge-a-package-for-automatic-evaluation-of-summaries/">https://www.microsoft.com/en-us/research/publication/rouge-a-package-for-automatic-evaluation-of-summaries/</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kishore</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Weinberger</surname><given-names>KQ</given-names> </name><name name-style="western"><surname>Artzi</surname><given-names>Y</given-names> </name></person-group><article-title>BERTScore: evaluating text generation with BERT</article-title><access-date>2026-07-04</access-date><conf-name>8th International Conference on Learning Representations Addis Ababa</conf-name><conf-date>Apr 26-30, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=SkeHuCVFDr">https://openreview.net/forum?id=SkeHuCVFDr</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bamberger</surname><given-names>R</given-names> </name><name name-style="western"><surname>Vanecek</surname><given-names>E</given-names> </name></person-group><source>Lesen-Verstehen-Lernen-Schreiben: Die Schwierigkeitsstufen von Texten in Deutscher Sprache [Book in German]</source><year>1984</year><publisher-name>Jugend und Volk</publisher-name><pub-id pub-id-type="other">978-3-425-01903-1</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ansch&#x00FC;tz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Groh</surname><given-names>G</given-names> </name></person-group><article-title>TUM social computing at GermEval 2022: towards the significance of text statistics and neural embeddings in text complexity prediction</article-title><access-date>2026-07-04</access-date><conf-name>Proceedings of the GermEval 2022 workshop on text complexity assessment of german text Potsdam</conf-name><conf-date>Sep 12-15, 2022</conf-date><conf-loc>Potsdam</conf-loc><fpage>21</fpage><lpage>26</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/anthology-files/anthology-files/pdf/germeval/2022.germeval-1.pdf">https://aclanthology.org/anthology-files/anthology-files/pdf/germeval/2022.germeval-1.pdf</ext-link></comment></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stetson</surname><given-names>PD</given-names> </name><name name-style="western"><surname>Bakken</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wrenn</surname><given-names>JO</given-names> </name><name name-style="western"><surname>Siegler</surname><given-names>EL</given-names> </name></person-group><article-title>Assessing electronic note quality using the Physician Documentation Quality Instrument (PDQI-9)</article-title><source>Appl Clin Inform</source><year>2012</year><volume>3</volume><issue>2</issue><fpage>164</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.4338/aci-2011-11-ra-0070</pub-id><pub-id pub-id-type="medline">22577483</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Fabbri</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Kry&#x015B;ci&#x0144;ski</surname><given-names>W</given-names> </name><name name-style="western"><surname>McCann</surname><given-names>B</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>C</given-names> </name><name name-style="western"><surname>Socher</surname><given-names>R</given-names> </name><name name-style="western"><surname>Radev</surname><given-names>D</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Roark</surname><given-names>B</given-names> </name><name name-style="western"><surname>Nenkova</surname><given-names>A</given-names> </name></person-group><article-title>SummEval: re-evaluating summarization evaluation</article-title><source>Transactions of the Association for Computational Linguistics</source><year>2021</year><volume>9</volume><publisher-name>MIT Press</publisher-name><fpage>391</fpage><lpage>409</lpage><pub-id pub-id-type="doi">10.1162/tacl_a_00373</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Maynez</surname><given-names>J</given-names> </name><name name-style="western"><surname>Narayan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bohnet</surname><given-names>B</given-names> </name><name name-style="western"><surname>McDonald</surname><given-names>R</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Jurafsky</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schluter</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tetreault</surname><given-names>J</given-names> </name></person-group><article-title>On faithfulness and factuality in abstractive summarization</article-title><conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>Jul 5-10, 2020</conf-date><fpage>1906</fpage><lpage>1919</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.173</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lommel</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Burchardt</surname><given-names>A</given-names> </name><name name-style="western"><surname>Uszkoreit</surname><given-names>H</given-names> </name></person-group><article-title>Multidimensional quality metrics: a flexible system for assessing translation quality</article-title><access-date>2026-07-04</access-date><conf-name>Proceedings of Translating and the Computer 35</conf-name><conf-date>Nov 28-29, 2013</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2013.tc-1.6.pdf">https://aclanthology.org/2013.tc-1.6.pdf</ext-link></comment></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Venkatesh</surname><given-names>V</given-names> </name><name name-style="western"><surname>Morris</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Davis</surname><given-names>GB</given-names> </name><name name-style="western"><surname>Davis</surname><given-names>FD</given-names> </name></person-group><article-title>User acceptance of information technology: toward a unified view</article-title><source>MIS Q</source><year>2003</year><month>09</month><day>1</day><volume>27</volume><issue>3</issue><fpage>425</fpage><lpage>478</lpage><pub-id pub-id-type="doi">10.2307/30036540</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Guerberof</surname><given-names>A</given-names> </name></person-group><article-title>Productivity and quality in MT post-editing</article-title><access-date>2026-07-04</access-date><conf-name>Beyond Translation Memories: New Tools for Translators Workshop</conf-name><conf-date>Aug 26-30, 2009</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2009.mtsummit-btm.7/">https://aclanthology.org/2009.mtsummit-btm.7/</ext-link></comment></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Scarton</surname><given-names>S</given-names> </name><name name-style="western"><surname>Forcada</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Espl&#x00E0;-Gomis</surname><given-names>M</given-names> </name><name name-style="western"><surname>Specia</surname><given-names>L</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Niehues</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cattoni</surname><given-names>R</given-names> </name><name name-style="western"><surname>St&#x00FC;ker</surname><given-names>S</given-names> </name><name name-style="western"><surname>Negri</surname><given-names>M</given-names> </name><name name-style="western"><surname>Turchi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ha</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Salesky</surname><given-names>E</given-names> </name><name name-style="western"><surname>Sanabria</surname><given-names>R</given-names> </name><name name-style="western"><surname>Barrault</surname><given-names>L</given-names> </name><name name-style="western"><surname>Specia</surname><given-names>L</given-names> </name><name name-style="western"><surname>Federico</surname><given-names>M</given-names> </name></person-group><article-title>Estimating post-editing effort: a study on human judgements, task-based and reference-based metrics of MT quality</article-title><year>2019</year><access-date>2026-07-04</access-date><conf-name>Proceedings of the 16th International Conference on Spoken Language Translation</conf-name><conf-date>Nov 2-3, 2019</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2019.iwslt-1.23.pdf">https://aclanthology.org/2019.iwslt-1.23.pdf</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reichheld</surname><given-names>FF</given-names> </name></person-group><article-title>The one number you need to grow</article-title><source>Harv Bus Rev</source><year>2003</year><month>12</month><volume>81</volume><issue>12</issue><fpage>46</fpage><lpage>54</lpage><pub-id pub-id-type="medline">14712543</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="web"><article-title>Streamlit/streamlit</article-title><source>GitHub</source><access-date>2026-07-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/streamlit/streamlit">https://github.com/streamlit/streamlit</ext-link></comment></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wilcoxon</surname><given-names>F</given-names> </name></person-group><article-title>Probability tables for individual comparisons by ranking methods</article-title><source>Biometrics</source><year>1947</year><month>09</month><volume>3</volume><issue>3</issue><fpage>119</fpage><pub-id pub-id-type="doi">10.2307/3001946</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Benjamini</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hochberg</surname><given-names>Y</given-names> </name></person-group><article-title>Controlling the false discovery rate: a practical and powerful approach to multiple testing</article-title><source>J R Stat Soc Ser B</source><year>1995</year><month>01</month><day>1</day><volume>57</volume><issue>1</issue><fpage>289</fpage><lpage>300</lpage><pub-id pub-id-type="doi">10.1111/j.2517-6161.1995.tb02031.x</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Conover</surname><given-names>WJ</given-names> </name><name name-style="western"><surname>Conover</surname><given-names>WJ</given-names> </name></person-group><source>Practical Nonparametric Statistics</source><year>1999</year><edition>3</edition><publisher-name>Wiley</publisher-name><pub-id pub-id-type="other">978-0-471-16068-7</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gwet</surname><given-names>KL</given-names> </name></person-group><article-title>Computing inter-rater reliability and its variance in the presence of high agreement</article-title><source>Br J Math Stat Psychol</source><year>2008</year><month>05</month><volume>61</volume><issue>Pt 1</issue><fpage>29</fpage><lpage>48</lpage><pub-id pub-id-type="doi">10.1348/000711006X126600</pub-id><pub-id pub-id-type="medline">18482474</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Walsh</surname><given-names>DP</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Buhl</surname><given-names>LK</given-names> </name><name name-style="western"><surname>Neves</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Mitchell</surname><given-names>JD</given-names> </name></person-group><article-title>Assessing interrater reliability of a faculty-provided feedback rating instrument</article-title><source>J Med Educ Curric Dev</source><year>2022</year><volume>9</volume><fpage>23821205221093205</fpage><pub-id pub-id-type="doi">10.1177/23821205221093205</pub-id><pub-id pub-id-type="medline">35677580</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kryscinski</surname><given-names>W</given-names> </name><name name-style="western"><surname>McCann</surname><given-names>B</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>C</given-names> </name><name name-style="western"><surname>Socher</surname><given-names>R</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Webber</surname><given-names>B</given-names> </name><name name-style="western"><surname>Cohn</surname><given-names>T</given-names> </name><name name-style="western"><surname>He</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name></person-group><article-title>Evaluating the factual consistency of abstractive text summarization</article-title><conf-name>Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)</conf-name><conf-date>Nov 16-20, 2020</conf-date><fpage>9332</fpage><lpage>9346</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.emnlp-main.750</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Min</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lyu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Holtzman</surname><given-names>A</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Goldberg</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kozareva</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name></person-group><article-title>Rethinking the role of demonstrations: what makes in-context learning work?</article-title><conf-name>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Dec 7-11, 2022</conf-date><conf-loc>Abu Dhabi, United Arab Emirates</conf-loc><fpage>11048</fpage><lpage>11064</lpage><pub-id pub-id-type="doi">10.18653/v1/2022.emnlp-main.759</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name></person-group><article-title>Comment on &#x201C;classifying the clinical significance of common breast pain symptoms using a large language model, ChatGPT (GPT-4)&#x201D;</article-title><source>Clin Imaging</source><year>2026</year><month>04</month><volume>132</volume><fpage>110741</fpage><pub-id pub-id-type="doi">10.1016/j.clinimag.2026.110741</pub-id><pub-id pub-id-type="medline">41671902</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Subramanian</surname><given-names>S</given-names> </name><name name-style="western"><surname>Elango</surname><given-names>V</given-names> </name><name name-style="western"><surname>Gungor</surname><given-names>M</given-names> </name></person-group><article-title>Small language models (SLMs) can still pack a punch: a survey</article-title><source>arXiv</source><comment>Preprint posted online on  May 14, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.05465</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goddard</surname><given-names>K</given-names> </name><name name-style="western"><surname>Roudsari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wyatt</surname><given-names>JC</given-names> </name></person-group><article-title>Automation bias: a systematic review of frequency, effect mediators, and mitigators</article-title><source>J Am Med Inf Assoc</source><year>2012</year><volume>19</volume><issue>1</issue><fpage>121</fpage><lpage>127</lpage><pub-id pub-id-type="doi">10.1136/amiajnl-2011-000089</pub-id><pub-id pub-id-type="medline">21685142</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name></person-group><article-title>Toward retrieval-grounded evaluation for conversational large language model-based risk assessment</article-title><source>JMIR AI</source><year>2026</year><month>03</month><day>12</day><volume>5</volume><fpage>e90759</fpage><pub-id pub-id-type="doi">10.2196/90759</pub-id><pub-id pub-id-type="medline">41818631</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="web"><source>KITE</source><access-date>2026-07-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://kite.ikim.nrw/">https://kite.ikim.nrw/</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Model configurations and inference parameters.</p><media xlink:href="jmir_v28i1e99136_app1.docx" xlink:title="DOCX File, 15 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Prompt templates and JSON schemas.</p><media xlink:href="jmir_v28i1e99136_app2.docx" xlink:title="DOCX File, 4102 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Detailed results, per case type analyses, agreement statistics, and additional figures.</p><media xlink:href="jmir_v28i1e99136_app3.docx" xlink:title="DOCX File, 3104 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Detailed critical error analysis.</p><media xlink:href="jmir_v28i1e99136_app4.xlsx" xlink:title="XLSX File, 16 KB"/></supplementary-material></app-group></back></article>