<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e91405</article-id><article-id pub-id-type="doi">10.2196/91405</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Large Language Models for Distress Rating in Korean Psycho-Oncology Interviews: Exploratory Clinician-Benchmarked Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Kim</surname><given-names>Jaehyun</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Son</surname><given-names>Kyung-Lak</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yeom</surname><given-names>Chan-Woo</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kim</surname><given-names>Won-Hyoung</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lee</surname><given-names>Sun Hyung</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Shin</surname><given-names>Joon Sung</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yang</surname><given-names>Hyunsun</given-names></name><degrees>MA</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yoo</surname><given-names>Daehun</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Hahm</surname><given-names>Bong-Jin</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff6">6</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Neuropsychiatry, Seoul National University Hospital</institution><addr-line>101, Daehak-ro</addr-line><addr-line>Jongno-gu</addr-line><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff2"><institution>Seoul Rhema Psychiatric Clinic</institution><addr-line>Gangnam-gu</addr-line><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff3"><institution>Department of Psychiatry, Eulji University Uijeongbu Eulji Medical Center</institution><addr-line>Uijeongbu</addr-line><addr-line>Gyeonggi-do</addr-line><country>Republic of Korea</country></aff><aff id="aff4"><institution>Department of Psychiatry, Inha University Hospital</institution><addr-line>Jung-gu</addr-line><addr-line>Incheon</addr-line><country>Republic of Korea</country></aff><aff id="aff5"><institution>Genesis Lab Inc</institution><addr-line>Jung-gu</addr-line><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff6"><institution>Medical Research Center, Institute of Human Behavioral Medicine, Seoul National University</institution><addr-line>Jongno-gu</addr-line><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>AL-Asadi</surname><given-names>Ali</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Liu</surname><given-names>Dario</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Shin</surname><given-names>Daun</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Bong-Jin Hahm, MD, PhD, Department of Neuropsychiatry, Seoul National University Hospital, 101, Daehak-ro, Jongno-gu, Seoul, 03080, Republic of Korea, 82 2-2072-2557; <email>hahmbj@gmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>9</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e91405</elocation-id><history><date date-type="received"><day>14</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>04</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>05</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Jaehyun Kim, Kyung-Lak Son, Chan-Woo Yeom, Won-Hyoung Kim, Sun Hyung Lee, Joon Sung Shin, Hyunsun Yang, Daehun Yoo, Bong-Jin Hahm. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 9.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e91405"/><abstract><sec><title>Background</title><p>Psychiatric distress is common among patients with cancer; yet, systematic interview-based screening remains difficult to scale in routine clinical care. Large language models (LLMs) have shown promise as scalable tools for mental health assessment, but most existing evidence is derived from clinician-authored records, translated text, or proxy data. The performance characteristics, error patterns, and explanatory behaviors of contemporary LLMs when applied to authentic, non-English psychiatric interviews remain insufficiently characterized.</p></sec><sec><title>Objective</title><p>This exploratory study evaluated how contemporary LLMs reproduced psycho-oncologists&#x2019; item-level symptom ratings from real-world Korean psycho-oncology interviews, focusing on concordance, directional bias, and clinician-adjudicated error characteristics.</p></sec><sec sec-type="methods"><title>Methods</title><p>Between April 2024 and May 2025, 101 adults receiving oncologic care in South Korea underwent semistructured interviews. Board-certified psycho-oncologists provided real-time, time-stamped ratings of 39 items. Generative pretrained transformer 4o (GPT-4o), Claude 3.5 Sonnet, and Gemini 2.5 Flash generated item scores and brief rationales using identical Korean zero-shot rubrics. Concordance between clinician ratings was evaluated using ordinal and binary screening metrics. A paired Wilcoxon test compared patient-level total symptom burden. Binary mismatches were clinically adjudicated as ambiguous or definite overestimation or underestimation with an 8-etiology taxonomy. Model-generated rationales were further meta-evaluated using GPT-5.4, Claude Sonnet 4.6, and Gemini Pro 3.1 across 4 dimensions: citation (use of quoted supporting statements), structure (logical organization of the rationale), mapping (consistency between the rationale and the assigned item rating), and expansion (degree of interpretive elaboration beyond the explicit transcript content). The associations between meta-evaluation results and absolute error were examined using cross-classified mixed-effects models.</p></sec><sec sec-type="results"><title>Results</title><p>Out of 101 participants, 88 (87.1%) were predominantly female and had breast cancer as the primary cancer type (n=70, 69.3%). Across 3931 item-level ratings, all models showed good agreement with clinicians (intraclass correlation coefficient: 0.816&#x2010;0.872), with GPT-4o showing the highest agreement. Claude 3.5 and Gemini 2.5 yielded significantly higher patient-level symptom burden (adjusted <italic>P</italic>&#x003C;.001 and adjusted <italic>P</italic>=.002, respectively), whereas GPT-4o did not (adjusted <italic>P</italic>=.15). Clinician adjudication attributed 97 of 248 (39.1%) GPT-4o mismatches, 88 of 301 (29.2%) Claude 3.5 mismatches, and 118 of 355 (33.2%) Gemini 2.5 mismatches to intrinsic ambiguity in patient speech. Among definite errors, misapplication of severity thresholds was the predominant mechanism across models. Exploratory mixed-effects models showed that higher expansion and lower mapping scores were associated with larger absolute errors. However, the association with expansion may reflect case difficulty or ambiguity rather than causation.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In this exploratory clinician-benchmarked evaluation of authentic interviews from a Korean psycho-oncology sample comprising predominantly women and patients with breast cancer, LLMs showed high aggregate concordance with psycho-oncologists&#x2019; item-level ratings while differing in their error profiles. These findings support further evaluation of LLM-based item-level symptom-rating approaches in psycho-oncology. Validation in larger, more diverse, and independent cohorts is needed.</p></sec></abstract><kwd-group><kwd>large language model</kwd><kwd>psycho-oncology</kwd><kwd>mental health</kwd><kwd>distress</kwd><kwd>patients with cancer</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Cancer is increasingly recognized as a &#x201C;human-crisis&#x201D; that profoundly disrupts psychological and emotional well-being [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Depression, anxiety, fatigue, and demoralization are highly prevalent throughout the illness trajectory and are associated with poor quality of life, reduced treatment adherence, and worse clinical outcomes [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. Consequently, early detection and management of psychiatric distress are critical for comprehensive oncologic care. However, systematic screening of psychological symptoms remains challenging in daily practice because of time constraints, competing priorities, and a shortage of trained specialists [<xref ref-type="bibr" rid="ref2">2</xref>].</p><p>A reliable assessment is indispensable; however, conventional assessment approaches are insufficient. Self-report questionnaires are constrained by comprehension and disclosure biases [<xref ref-type="bibr" rid="ref4">4</xref>], and psychiatrist interviews, while the gold standard, are time-intensive and impractical for large-scale implementation [<xref ref-type="bibr" rid="ref5">5</xref>]. These challenges have motivated the exploration of technological solutions to complement clinical expertise.</p><p>In recent years, large language models (LLMs) have emerged as potential adjuncts to clinical expertise [<xref ref-type="bibr" rid="ref6">6</xref>]. With rapid advances in natural language processing, these models have demonstrated promise in medical contexts, including summarizing clinical records, extracting structured information from free-text notes, and supporting decision-making in various domains [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>Their application in mental health assessment, however, remains in an early and uniquely challenging phase [<xref ref-type="bibr" rid="ref10">10</xref>]. Unlike other areas of medicine where diagnosis and management rely primarily on explicit clinical findings, psychiatric assessment is inherently interpretive, requiring sensitivity to implicit cues, affective nuances, and contextual meanings [<xref ref-type="bibr" rid="ref11">11</xref>]. Such nuances are embedded not only in nonverbal cues but also within verbal narratives through indirect formulation and contextual framing [<xref ref-type="bibr" rid="ref12">12</xref>]. These subtle features introduce the risk of misclassification if AI systems are deployed without a demonstrated capacity to interpret such depth [<xref ref-type="bibr" rid="ref13">13</xref>]. Moreover, since most pretraining data and clinical benchmarks are English-dominant and culturally Western, LLM performance may not be generalizable to languages with different pragmatic conventions and symptom-expression norms [<xref ref-type="bibr" rid="ref14">14</xref>]. Consistent with this concern, a recent study reported higher hallucination rates and lower diagnostic accuracy when models were given Korean clinical text inputs compared to English translations, highlighting the need for language- and culture-specific validation before clinical deployment [<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>Furthermore, most LLM applications in medicine rely on clinician-authored records or other proxy texts rather than raw patient speech [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. Studies based on authentic psychiatrist-patient interactions, particularly full clinical interviews, therefore, remain limited [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Addressing this gap is important, as clinical interviews remain the cornerstone of psychiatric diagnosis by directly capturing patients&#x2019; lived experiences. Although LLMs are capable of linguistic processing, their reliability in capturing the implicit aspects of psychiatric discourse remains to be validated.</p><p>Psycho-oncology is a particularly compelling context for such research. Patients with cancer face psychiatric distress shaped by existential threats, physical suffering, and complex treatment regimens [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Although the high prevalence of depression, anxiety, and fatigue in this population is well established, technological methods for systematically evaluating these symptoms have not been rigorously tested [<xref ref-type="bibr" rid="ref2">2</xref>]. Few studies have validated these results with the judgments of psychiatrists with expertise in psycho-oncology [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. The feasibility and limitations of LLM-based tools in this population remain largely unknown.</p><p>This study aimed to evaluate the concordance between contemporary LLMs and board-certified psycho-oncologists in rating symptoms from semistructured interviews with patients receiving oncologic care. The study further aimed to characterize systematic patterns of disagreement and examine how the qualitative features of model reasoning are related to rating errors. Using a dataset of routine psycho-oncology interviews with time-stamped clinician ratings across 39 symptoms and interference items, three LLMs (generative pretrained transformer [GPT]-4o, Claude 3.5, and Gemini 2.5) were evaluated using a standardized zero-shot rubric. By integrating authentic patient-psychiatrist dialogue with item-level reference ratings and a structured analysis of disagreement mechanisms, this study could address a critical gap in the evaluation of LLMs for psychiatric assessment in real-world non-English clinical settings.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Ethical Considerations</title><p>This study was approved by the Institutional Review Board of Seoul National University Hospital (approval number 2401-084-1501) and was conducted in accordance with the Declaration of Helsinki. Written informed consent was obtained from all participants.</p></sec><sec id="s2-2"><title>Participants</title><p>From April 2024 to May 2025, adults (&#x2265;18 y) receiving oncologic care were recruited consecutively from outpatient and inpatient oncology services at 3 secondary and tertiary hospitals in South Korea. The exclusion criteria were medical, cognitive, or linguistic limitations that precluded meaningful interviews.</p></sec><sec id="s2-3"><title>Data Collection and Interview Procedure</title><p>Interviews were conducted by 3 psycho-oncology psychiatrists using a custom application that enabled real-time symptom scoring and time-stamped tagging. During the interviews, psychiatrists assigned ratings in real time using the application, with each entry automatically time-stamped. This design created a direct link between the clinical judgment and the corresponding segment of the audio recording, producing an annotated dataset that aligned symptom scores with a conversational context.</p><p>The interviews were conducted in a semistructured format, comprising 39 evaluation items across four domains: (1) psychiatric symptoms (29 items), (2) physical symptoms (4 items), (3) psychiatric interference (3 items), and (4) physical interference (3 items). Each item was rated on a 4-point Likert scale (0&#x2010;3), with higher scores indicating greater severity or interference.</p><p>All interview recordings were transcribed using speech-to-text (STT) software, manually corrected, and time-aligned by trained assistants without a medical background. All data were fully deidentified before the analysis.</p><p>To assess clinician reference rating reliability, an independent board-certified psychiatrist, blinded to the original ratings, reevaluated 2 complementary samples: 15 high-distress cases (&#x2265;12 clinically significant items) and an additional 15 cases sampled without restriction on distress severity. Interrater agreement was high in both the high-distress sample (intraclass correlation coefficient [ICC]=0.904) and the broader-severity sample (ICC=0.899). Agreement was also high in the pooled 30-case analysis (ICC=0.911), supporting the use of these ratings as a reference standard [<xref ref-type="bibr" rid="ref20">20</xref>].</p></sec><sec id="s2-4"><title>AI Evaluation</title><p>An LLM-based evaluation was conducted from June to August 2025 using APIs. Three LLMs were evaluated: GPT-4o (OpenAI), Claude 3.5 Sonnet (Anthropic), and Gemini 2.5 Flash (Google DeepMind), with a temperature of 0.3, a maximum token limit of 20,000, and default parameters otherwise.</p><p>All interview recordings, transcripts, clinician ratings, and adjudication data remained nonpublic throughout the study period and were not publicly released, posted online, deposited in public repositories, included in preprints, or shared as publicly accessible materials before or during the LLM evaluation. Only deidentified transcripts and item-specific evaluation inputs were used for API-based LLM inference. Therefore, contamination through publicly available training materials was considered highly unlikely.</p><p>For each patient-model combination, the complete Korean transcript and corresponding time-stamped item-specific windows were submitted in a single API call. The models were instructed to assign ordinal scores (0&#x2010;3) to each of the 39 items and generate brief textual justifications for each rating. If a call timed out or did not return a complete response conforming to the prespecified JSON schema, the same request was repeated, and the first complete schema-conforming response was retained. No additional generation was performed for that patient-model combination.</p><p>All models were guided by an identical zero-shot prompt in Korean that combined explicit evaluation rubrics with schema-constrained outputs and reasoning justifications. The prompt assigned models the role of a psychiatrist specializing in psycho-oncology. The instructions specified a fixed JSON structure with predefined keys for all items and required ratings on a standardized 4-point rubric anchored to symptom frequency, duration, and functional impairment over the preceding 1 to 2 weeks. The distinction between scores of 0 to 1 (subclinical) and 2 to 3 (clinically significant) was emphasized as the threshold for clinical significance. Conservative scoring was operationalized within the prompt by requiring clear transcript evidence for clinically significant ratings and by instructing models to assign lower scores when evidence regarding symptom frequency, duration, or functional impact was unclear or insufficient. The prompt also included interpretation guidelines to account for potential STT errors, such as phonetic misrecognitions, fragmented speech, and indirect emotional expressions.</p><p>During prompt development, we evaluated 2 rubric formats. One format used a single standardized 4-point rubric applied uniformly across all symptom items, whereas the other provided symptom-specific descriptions for scores 0, 1, 2, and 3 for each item. The standardized rubric format was selected for the final evaluation because pilot review showed more consistent application of shared scoring criteria across items, including symptom frequency, duration, and functional impact. We then compared 2 input-context formats. In the first, the model received only the transcript excerpt corresponding to each tagged symptom. In the second, the model received the complete interview transcript together with the item-specific excerpt (ie, the portion of the interview where the symptom was mentioned). The latter format was selected to preserve the broader clinical context of the interview.</p><p>These development steps concerned the general rubric and contextual input structure rather than model-specific prompt tuning. The final prompt was fixed before the main evaluation and applied identically across all models. No model-specific prompt modifications or few-shot demonstrations containing example transcripts, completed item ratings, or model-output examples were used. The only examples included in the prompt were generic anchor expressions embedded in the scoring rubric. An English translation of the full prompt is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-5"><title>Clinician-Adjudicated Error Analysis</title><p>Binary mismatches (clinically significant symptoms defined as score &#x2265;2) were manually reviewed and classified using a taxonomy developed and refined through consensus within a multidisciplinary team that included clinical experts and data scientists. Selected cases were reviewed in multidisciplinary team meetings for calibration of the operational definitions and consistency of taxonomy application. A single psychiatrist served as the primary adjudicator and performed the case-level classification with explicit model-source labels masked. For each mismatch, the adjudicator reviewed the target item, the deidentified transcript and item-specific segment, the clinician and model ratings, and the model-generated rationale.</p><p>Disagreements were categorized as ambiguous (insufficient or equivocal transcript evidence) or definite overestimation or underestimation, with directional errors further classified into 8 etiologies (<xref ref-type="table" rid="table1">Table 1</xref>). Reproducibility was assessed through an independent, model-blinded review of 150 mismatches stratified by model and initial adjudication category, yielding high interrater agreement for the ambiguous-versus-definite classification (Cohen &#x03BA;=0.82). To provide a clearer understanding of ambiguous disagreements, 3 deidentified and translated examples are presented in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Clinician-adjudicated error taxonomy for large language model-clinician discrepancies.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Subtype</td><td align="left" valign="bottom">Definition</td></tr></thead><tbody><tr><td align="left" valign="top">Implicit cues</td><td align="left" valign="top">Misinterpretation of subtle or indirect emotional cues, either by overattributing or underrecognizing clinically meaningful affective signals.</td></tr><tr><td align="left" valign="top">Other information</td><td align="left" valign="top">Use of information drawn from the wrong symptom domain, reflecting misalignment between the evaluated item and the evidence selected from the transcript.</td></tr><tr><td align="left" valign="top">Comprehension</td><td align="left" valign="top">Errors arising from misunderstanding, mis-parsing, or insufficiently attending to the literal content of the patient&#x2019;s statements.</td></tr><tr><td align="left" valign="top">Criteria application</td><td align="left" valign="top">Incorrect application of standardized scoring criteria, including misjudgment of symptom frequency, duration, or functional impact.</td></tr><tr><td align="left" valign="top">Personality attribution</td><td align="left" valign="top">Confusion between stable personality characteristics and acute clinical symptoms, resulting in distorted severity assessment.</td></tr><tr><td align="left" valign="top">Somatic confounding</td><td align="left" valign="top">Misattribution between physical and psychiatric symptoms, leading to overestimation or underestimation of psychological distress.</td></tr><tr><td align="left" valign="top">Temporal misalignment</td><td align="left" valign="top">Incorrect interpretation of the temporal reference window, including misclassification of symptom timing relative to the 1- to 2-week evaluation frame.</td></tr><tr><td align="left" valign="top">Construct misclassification</td><td align="left" valign="top">Incorrect mapping of a symptom expression to the wrong clinical construct, reflecting confusion in symptom definition rather than mis-selection of transcript evidence.</td></tr></tbody></table></table-wrap></sec><sec id="s2-6"><title>Automated Meta-Evaluation of Reasoning</title><p>To systematically investigate the qualitative characteristics of the LLM outputs, an automated meta-evaluation framework was applied to all model-generated responses between June and July 2026. To reduce dependence on any single evaluator model family, model-generated rationales were independently meta-evaluated by GPT-5.4, Claude Sonnet 4.6, and Gemini Pro 3.1 using the same rubric, with explicit source-model identifiers removed. Reasoning outputs were assessed across 4 dimensions: citation (specificity of referenced evidence), structure (evidence, interpretation, and score completeness), mapping (rules and consistency between interpretation and score), and expansion (extrapolation beyond the transcript). For each rationale and dimension, the arithmetic mean of the 3 independently assigned scores was used as the meta-evaluation score. In a 10-patient subset, agreement between the mean meta-evaluation scores and psychiatrist ratings ranged from ICC (A,1)=0.646 to 0.822 across dimensions. Detailed reliability methods and results are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>The meta-evaluation prompt was developed independently of the primary symptom-rating prompt. During prompt development, the rubric was refined through pilot review to improve interpretability and reduce common scoring artifacts, including over-rewarding verbose rationales, penalizing appropriate documentation of symptom absence, or relying on keyword-based scoring. After pilot review, the prompt was finalized and applied uniformly to the full set of model-generated rationales. The parameters were a temperature of 0.15 and a maximum token count of 131,702. An English translation of the full prompt is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-7"><title>Statistical Analysis</title><p>Statistical analyses were performed using Python (version 3.12; libraries: NumPy, SciPy, pandas, scikit-learn, and statsmodels). All hypothesis tests were 2-sided, with the threshold for statistical significance set at <italic>P</italic>&#x003C;.05. To address the multiplicity of related tests (patient-level directional bias, meta-feature comparisons, and mixed-effects models), <italic>P</italic> values were adjusted using the Benjamini-Hochberg false discovery rate. For 2-df Wald tests comparing source-model-specific ICCs across the 4 meta-evaluation dimensions and the post hoc ambiguity analyses, the Holm procedure was applied.</p><p>Because this study was designed to estimate model agreement, directional bias, and error patterns rather than to test a single confirmatory hypothesis, all consecutively recruited eligible participants during the study period were analyzed. Although each interview contributed up to 39 item-level evaluations, these observations were clustered within patients and were not treated as independent patient-level cases. To assess the stability of key performance estimates, we calculated 95% CIs using nonparametric patient-level cluster bootstrapping with 2000 resamples. In each bootstrap resample, patients rather than individual item-level observations were sampled with replacement, and all item-level observations from each selected patient were retained.</p><p>Descriptive statistics were used to summarize the participant demographics.</p><p>The concordance between LLM outputs and psychiatrist ratings on a 0 to 3 ordinal scale was assessed using quadratic weighted kappa (QWK) and ICC (2,1). The exact accuracy was also reported. For screening performance, scores were dichotomized (0&#x2010;1 vs &#x2265;2) to compute sensitivity, specificity, positive predictive value (PPV), negative predictive value (NPV), and <italic>F</italic><sub>1</sub>-score. As a sensitivity analysis, the 170 patient-item pairs classified as ambiguous for any of the 3 models were excluded identically from all models, and the binary screening metrics were recalculated. The patient-level directional bias was evaluated by comparing the total clinically significant symptom counts per patient using the paired Wilcoxon signed-rank test.</p><p>Post hoc exploratory analyses of ambiguous mismatches were performed to examine whether the proportion of adjudicated mismatches classified as ambiguous differed across models or clinical domains. A total of 904 binary mismatches were analyzed using patient-clustered logistic generalized estimating equation models, which accounted for multiple mismatch observations from the same patient. Ambiguity status (ambiguous vs definite error) was the outcome, and model and clinical domain (psychiatric symptoms, physical symptoms, psychiatric interference, and physical interference) were included simultaneously as predictors. Following the global model test, pairwise model contrasts were estimated. To examine whether model differences varied across clinical domains, we tested the model-by-domain interaction. To examine whether model differences varied according to patient characteristics, separate domain-adjusted models tested interactions between model and standardized age, standardized clinician-rated symptom burden, major depressive episode status, and sex.</p><p>To examine whether reasoning styles differed across the models, the distributions of the meta-evaluation scores were compared using the Friedman test. To evaluate associations between rationale characteristics and prediction error while accounting for cross-classified data structures, mixed-effects models were fitted with the absolute error as a dependent variable. Meta-evaluation scores were included as fixed effects and were <italic>z</italic>-standardized within each model. Clinical domain (psychiatric symptoms, physical symptoms, psychiatric interference, and physical interference) was included as a 4-level fixed effect to account for systematic differences in error across domains. Random intercepts were specified for patients and items to account for within-patient clustering and item-level heterogeneity, respectively. The models were estimated using restricted maximum likelihood. As a sensitivity analysis, the same model specification was applied to the 10-patient subset using the human reference meta-evaluation scores. The resulting coefficients were compared with those obtained using the mean scores assigned by the 3 evaluator models in the same 10-patient subset and in the full cohort.</p></sec><sec id="s2-8"><title>Reporting Guideline</title><p>The TRIPOD-LLM (Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis&#x2013;Large Language Model) reporting guideline was used to guide the reporting of this study. The completed checklist is provided in <xref ref-type="supplementary-material" rid="app5">Checklist 1</xref> [<xref ref-type="bibr" rid="ref21">21</xref>].</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Participant Characteristics</title><p>This study included 101 patients receiving oncology treatment or follow-up care. Each interview yielded up to 39 symptom-level evaluations. Eight symptom tags were excluded because of incomplete or corrupted audio recordings that precluded reliable transcription. The final dataset comprised 3931 symptom evaluations. Demographic characteristics of participants are shown in <xref ref-type="table" rid="table2">Table 2</xref>. The mean age of participants was 52.8 (SD 9.2) years, and the majority (n=88, 87.1%) were female participants. The most common primary cancer was breast cancer (n=70, 69.3%). Most participants (n=85, 84.2%) were receiving treatment after initial diagnosis or recurrence, while a smaller proportion were in posttreatment follow-up, remission follow-up, palliative care, or had unknown treatment status. Major depressive episodes were identified in 17.8% (n=18) of the participants. The distribution of clinician-rated symptom scores was skewed toward lower severity.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Demographic characteristics of participants.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Demographic characteristics</td><td align="left" valign="bottom">Values, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Sex</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">13 (12.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">88 (87.1)</td></tr><tr><td align="left" valign="top" colspan="2">Age (y)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>21-30</td><td align="left" valign="top">1 (1.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>31-40</td><td align="left" valign="top">9 (8.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>41-50</td><td align="left" valign="top">36 (35.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>51-60</td><td align="left" valign="top">33 (32.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>61-70</td><td align="left" valign="top">19 (18.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>71-80</td><td align="left" valign="top">3 (3.0)</td></tr><tr><td align="left" valign="top" colspan="2">Cancer type</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Breast</td><td align="left" valign="top">70 (69.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Colorectal</td><td align="left" valign="top">9 (8.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Lung</td><td align="left" valign="top">5 (5.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Stomach</td><td align="left" valign="top">4 (4.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pancreatic</td><td align="left" valign="top">3 (3.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ovarian</td><td align="left" valign="top">2 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Others</td><td align="left" valign="top">8 (8.0)</td></tr><tr><td align="left" valign="top" colspan="2">Treatment status</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Initial treatment</td><td align="left" valign="top">69 (68.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Posttreatment surveillance</td><td align="left" valign="top">9 (8.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Treatment for recurrence</td><td align="left" valign="top">16 (15.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Remission follow-up</td><td align="left" valign="top">1 (1.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Palliative care</td><td align="left" valign="top">2 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Unknown</td><td align="left" valign="top">4 (4.0)</td></tr><tr><td align="left" valign="top" colspan="2">MDE<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mild</td><td align="left" valign="top">3 (3.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Moderate</td><td align="left" valign="top">13 (12.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Severe</td><td align="left" valign="top">2 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Total</td><td align="char" char="." valign="top">18 (17.8)</td></tr><tr><td align="left" valign="top" colspan="2">Symptom score</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0</td><td align="left" valign="top">2106 (53.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>1</td><td align="left" valign="top">1082 (27.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>2</td><td align="left" valign="top">657 (16.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>3</td><td align="left" valign="top">86 (2.2)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>MDE: major depressive episode.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Primary Diagnostic Performance</title><p>Across all 3 models, the concordance with board-certified psychiatrists on the 0 to 3 ordinal scale was high (<xref ref-type="table" rid="table3">Table 3</xref> and Figure S1 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). GPT-4o showed the highest agreement (ICC=0.872), followed by Claude 3.5 (ICC=0.838) and Gemini 2.5 (ICC=0.816).</p><p>When scores were dichotomized at &#x2265;2 to indicate clinically significant items, all models demonstrated high performance (<xref ref-type="table" rid="table3">Table 3</xref> and <xref ref-type="fig" rid="figure1">Figure 1</xref>). GPT-4o showed the highest <italic>F</italic><sub>1</sub>-score (0.837), followed by Claude 3.5 (0.813) and Gemini 2.5 (0.775). In the sensitivity analysis excluding ambiguous mismatches, <italic>F</italic><sub>1</sub>-scores were numerically higher for all 3 models, while the model ordering remained unchanged (Table S1 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Symptom rating performance of large language models, including item-level exact agreement (ordinal accuracy), ordinal reliability metrics, and binary screening performance<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Evaluation measure</td><td align="left" valign="bottom">GPT-4o (95% CI)</td><td align="left" valign="bottom">Claude 3.5 (95% CI)</td><td align="left" valign="bottom">Gemini 2.5 (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Ordinal</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Accuracy</td><td align="left" valign="top">0.843 (0.823&#x2010;0.862)</td><td align="left" valign="top">0.797 (0.770&#x2010;0.824)</td><td align="left" valign="top">0.794 (0.769&#x2010;0.819)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ICC<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="top">0.872 (0.857&#x2010;0.886)</td><td align="left" valign="top">0.838 (0.818&#x2010;0.853)</td><td align="left" valign="top">0.816 (0.792&#x2010;0.837)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>QWK<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">0.872 (0.857&#x2010;0.886)</td><td align="left" valign="top">0.838 (0.818&#x2010;0.853)</td><td align="left" valign="top">0.816 (0.792&#x2010;0.837)</td></tr><tr><td align="left" valign="top" colspan="4">Binary</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sensitivity</td><td align="left" valign="top">0.857 (0.821&#x2010;0.890)</td><td align="left" valign="top">0.879 (0.844&#x2010;0.909)</td><td align="left" valign="top">0.825 (0.782&#x2010;0.861)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Specificity</td><td align="left" valign="top">0.955 (0.943&#x2010;0.966)</td><td align="left" valign="top">0.934 (0.917&#x2010;0.949)</td><td align="left" valign="top">0.929 (0.912&#x2010;0.943)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>PPV<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">0.818 (0.771&#x2010;0.856)</td><td align="left" valign="top">0.756 (0.699&#x2010;0.804)</td><td align="left" valign="top">0.732 (0.666&#x2010;0.784)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>NPV<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup></td><td align="left" valign="top">0.966 (0.958&#x2010;0.974)</td><td align="left" valign="top">0.971 (0.962&#x2010;0.978)</td><td align="left" valign="top">0.958 (0.945&#x2010;0.969)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">0.837 (0.805&#x2010;0.864)</td><td align="left" valign="top">0.813 (0.775&#x2010;0.844)</td><td align="left" valign="top">0.775 (0.731&#x2010;0.813)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Adjusted <italic>P</italic> value</td><td align="left" valign="top">.15</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">.002</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Values are point estimates with patient-cluster bootstrap. Ordinal metrics were calculated on the original 4-point scale (0&#x2010;3). Binary metrics were computed using a clinical threshold of &#x2265;2. Intraclass correlation coefficient (2,1) and quadratic weighted kappa were calculated separately; their identical values to 3 decimal places reflect rounding. Adjusted <italic>P</italic> values indicate Benjamini-Hochberg false discovery rate adjusted <italic>P</italic> values from paired comparisons of patient-level total symptom burden (sum of binary flags) between each model and clinician ratings.</p></fn><fn id="table3fn2"><p><sup>b</sup>ICC: intraclass correlation coefficient.</p></fn><fn id="table3fn3"><p><sup>c</sup>QWK: quadratic weighted kappa.</p></fn><fn id="table3fn4"><p><sup>d</sup>PPV: positive predictive value.</p></fn><fn id="table3fn5"><p><sup>e</sup>NPV: negative predictive value.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Binary confusion matrices for GPT-4o, Claude 3.5, and Gemini 2.5 Binary confusion matrices compare each model&#x2019;s dichotomized predictions (0: score &#x003C;2, clinically absent or subclinical; 1: score &#x2265;2, clinically significant) with clinician ratings across 3931 symptom evaluations. The cells indicate true negatives, false positives, false negatives, and true positives for each model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91405_fig01.png"/></fig></sec><sec id="s3-3"><title>Systematic Directional Bias in Symptom Ratings</title><p>Directional error patterns for binary classifications (score &#x2265;2 vs &#x003C;2) are summarized in <xref ref-type="fig" rid="figure1">Figure 1</xref>. Across all models, overestimation errors (LLMs: &#x2265;2, clinicians: &#x003C;2) were more frequent than underestimation errors (LLMs: &#x003C;2, clinicians: &#x2265;2), but the relative proportions differed by model. Overestimation accounted for 142 of 248 (57.3%) errors for GPT-4o, 211 of 301 (70.1%) errors for Claude 3.5, and 225 of 355 (63.4%) errors for Gemini 2.5 (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><p>Paired Wilcoxon signed-rank tests comparing the total symptom burden between each LLM and the clinician reference standard showed no statistically significant difference for GPT-4o (adjusted <italic>P</italic>=.15), whereas Claude 3.5 (adjusted <italic>P</italic>&#x003C;.001) and Gemini 2.5 (adjusted <italic>P</italic>=.002) yielded higher total burdens than the clinicians (<xref ref-type="table" rid="table3">Table 3</xref>).</p></sec><sec id="s3-4"><title>Clinician-Adjudicated Error Taxonomy</title><p>All binary mismatches between LLM and clinician ratings were classified by the primary adjudicator using the consensus-developed error taxonomy (<xref ref-type="table" rid="table1">Table 1</xref>), and the resulting classification frequencies are summarized in <xref ref-type="table" rid="table4">Table 4</xref><italic>.</italic> Ambiguous cases accounted for a substantial proportion of disagreements across all models, comprising 88 of 301 (29.2%) mismatches for Claude 3.5, 118 of 355 (33.2%) mismatches for Gemini 2.5, and 97 of 248 (39.1%) mismatches for GPT-4o. A global difference across models was detected (Wald <italic>&#x03C7;</italic>&#x00B2;<sub>&#x2082;</sub>=6.58; <italic>P</italic>=.04). In the Holm-adjusted pairwise comparisons, GPT-4o had higher odds of ambiguity than Claude 3.5 (adjusted <italic>P</italic>=.03), whereas the other model contrasts were not significant (Table S3 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). Neither the clinical-domain effect nor the model-by-domain interaction was statistically significant. No patient-characteristic main effect remained significant after Holm adjustment, and no model interaction was detected for age, clinician-rated item burden, major depressive episode status, or sex (all interaction-adjusted <italic>P</italic>&#x003E;.99; Table S4 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p><p>Among all model-specific mismatches, nonambiguous overestimation and underestimation accounted for 145 of 301 (48.2%) and 68 of 301 (22.6%), respectively, for Claude 3.5; 141 of 355 (39.7%) and 96 of 355 (27.0%) for Gemini 2.5; and 73 of 248 (29.4%) and 78 of 248 (31.5%) for GPT-4o (<xref ref-type="table" rid="table4">Table 4</xref>). Thus, even after excluding ambiguous cases, overestimation errors remained relatively more frequent than underestimation errors for Claude 3.5 and Gemini 2.5, whereas GPT-4o showed a more balanced distribution.</p><p><xref ref-type="fig" rid="figure2">Figure 2</xref> shows the composition of nonambiguous errors by subtype, displaying the model-wise distributions after renormalizing the proportions within each model and excluding ambiguous cases. Criteria-related errors were the most frequent nonambiguous subtype across all 3 models. Claude 3.5 and Gemini 2.5 predominantly overestimated severity based on misapplied criteria, whereas GPT-4o showed a more balanced profile with relatively higher criteria-related underestimation. Implicit cue errors accounted for smaller proportions in GPT-4o than in Claude 3.5 and Gemini 2.5. Less frequent subtypes, such as errors involving other information, somatic confounding, temporal misalignment, comprehension problems, and personality attribution, each accounted for only a small proportion of nonambiguous mismatches in all models.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Distribution of clinician-adjudicated error types across large language models, including ambiguous cases and direction-specific misclassifications<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Direction and error type</td><td align="left" valign="bottom">GPT-4o, n (%)</td><td align="left" valign="bottom">Claude 3.5, n (%)</td><td align="left" valign="bottom">Gemini 2.5, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Ambiguous</td><td align="left" valign="top">97 (39.1)</td><td align="left" valign="top">88 (29.2)</td><td align="left" valign="top">118 (33.2)</td></tr><tr><td align="left" valign="top" colspan="4">Over</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Criteria<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">35 (14.1)</td><td align="left" valign="top">76 (25.2)</td><td align="left" valign="top">81 (22.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Implicit cues</td><td align="left" valign="top">9 (3.6)</td><td align="left" valign="top">25 (8.3)</td><td align="left" valign="top">17 (4.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Comprehension</td><td align="left" valign="top">3 (1.2)</td><td align="left" valign="top">7 (2.3)</td><td align="left" valign="top">7 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other information</td><td align="left" valign="top">11 (4.4)</td><td align="left" valign="top">13 (4.3)</td><td align="left" valign="top">6 (1.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Personality<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">4 (1.6)</td><td align="left" valign="top">5 (1.7)</td><td align="left" valign="top">4 (1.1)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Somatic<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">7 (2.8)</td><td align="left" valign="top">11 (3.7)</td><td align="left" valign="top">12 (3.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Temporal<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td><td align="left" valign="top">3 (1.2)</td><td align="left" valign="top">7 (2.3)</td><td align="left" valign="top">13 (3.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Construct<sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup></td><td align="left" valign="top">1 (0.4)</td><td align="left" valign="top">1 (0.3)</td><td align="left" valign="top">1 (0.3)</td></tr><tr><td align="left" valign="top" colspan="4">Under</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Criteria</td><td align="left" valign="top">41 (16.5)</td><td align="left" valign="top">13 (4.3)</td><td align="left" valign="top">13 (3.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Implicit cues</td><td align="left" valign="top">23 (9.3)</td><td align="left" valign="top">44 (14.6)</td><td align="left" valign="top">71 (20.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Comprehension</td><td align="left" valign="top">2 (0.8)</td><td align="left" valign="top">2 (0.7)</td><td align="left" valign="top">3 (0.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other information</td><td align="left" valign="top">5 (2.0)</td><td align="left" valign="top">2 (0.7)</td><td align="left" valign="top">1 (0.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Personality</td><td align="left" valign="top">0 (0.0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Somatic</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Temporal</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Construct</td><td align="left" valign="top">7 (2.8)</td><td align="left" valign="top">7 (2.3)</td><td align="left" valign="top">8 (2.3)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Values are presented as number and percentage of all binary mismatches for each model. Percentages are calculated within each model using the total number of mismatches as the denominator, including ambiguous cases. Ambiguous cases indicate instances in which the expert adjudicator judged the transcript insufficient to determine a definite overestimation or underestimation.</p></fn><fn id="table4fn2"><p><sup>b</sup>Criteria: criteria application.</p></fn><fn id="table4fn3"><p><sup>c</sup>Personality: personality attribution.</p></fn><fn id="table4fn4"><p><sup>d</sup>Somatic: somatic confounding.</p></fn><fn id="table4fn5"><p><sup>e</sup>Temporal: temporal misalignment.</p></fn><fn id="table4fn6"><p><sup>f</sup>Construct: construct misapplication.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Model-wise distribution of nonambiguous error subtypes. Radar plots showing the distribution of clinician-adjudicated error subtypes among nonambiguous mismatches for GPT-4o, Claude 3.5, and Gemini 2.5. For each model, subtype proportions were renormalized to a sum of 1 after excluding ambiguous cases. The right semicircle represents overestimation subtypes, and the left semicircle represents underestimation subtypes (implicit cues, other information, comprehension, criteria application, personality attribution, somatic confounding, temporal misalignment, and construct misclassification).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91405_fig02.png"/></fig></sec><sec id="s3-5"><title>Reasoning Style and Association With Prediction Error</title><p>Meta-evaluation scores differed significantly between the models across all 4 dimensions (citation, structure, mapping, expansion; all <italic>P</italic>&#x003C;.001; <xref ref-type="table" rid="table5">Table 5</xref>). Claude 3.5 had the highest citation and structure scores, Gemini 2.5 had intermediate scores, and GPT-4o had the lowest scores on these dimensions. Mapping scores were similar for Claude 3.5 and Gemini 2.5 and lower for GPT-4o, whereas expansion scores were similar for GPT-4o and Claude 3.5 and lower for Gemini 2.5 (<xref ref-type="table" rid="table5">Table 5</xref>).</p><p>In exploratory domain-adjusted cross-classified mixed-effects models, mapping was inversely associated with absolute error, whereas structure and expansion were positively associated with error across all 3 source models. Citation was positively associated with error for GPT-4o and Claude 3.5 but not for Gemini 2.5 (Table S5 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). In the human-reference sensitivity analyses, coefficient directions were consistent with those observed in the primary analysis for all 4 dimensions, although statistical significance varied across dimensions and source models (Figure S2 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>).</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Meta-evaluation scores of model-generated rationales across 4 explanatory dimensions for each large language model<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimensions</td><td align="left" valign="bottom">GPT-4o</td><td align="left" valign="bottom">Claude 3.5</td><td align="left" valign="bottom">Gemini 2.5</td><td align="left" valign="bottom">Adjusted <italic>P</italic> value</td><td align="left" valign="bottom">Kendall&#x2019;s <italic>W</italic></td></tr></thead><tbody><tr><td align="left" valign="top">Citation</td><td align="left" valign="top">4.304</td><td align="left" valign="top">4.705</td><td align="left" valign="top">4.636</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.762</td></tr><tr><td align="left" valign="top">Structure</td><td align="left" valign="top">3.991</td><td align="left" valign="top">4.492</td><td align="left" valign="top">4.324</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.812</td></tr><tr><td align="left" valign="top">Mapping</td><td align="left" valign="top">4.066</td><td align="left" valign="top">4.295</td><td align="left" valign="top">4.293</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.540</td></tr><tr><td align="left" valign="top">Expansion</td><td align="left" valign="top">1.785</td><td align="left" valign="top">1.787</td><td align="left" valign="top">1.709</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.166</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Scores represent mean ratings on a 1 to 5 ordinal scale, with higher values indicating a greater presence of the corresponding feature. Adjusted <italic>P</italic> values indicate false discovery rate&#x2013;adjusted <italic>P</italic> values for overall Friedman test results. Kendall&#x2019;s <italic>W</italic> represents the effect size associated with the Friedman test.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><p>This exploratory study examined how 3 LLMs reproduced psycho-oncologists&#x2019; item-level ratings from real Korean psycho-oncology interviews and how their reasoning patterns were related to errors. Across the 3 evaluated models, overall concordance with clinician ratings was high, suggesting that these models could approximate psycho-oncologists&#x2019; item-level ratings at the aggregate level. However, the structure of disagreement differed meaningfully between models. Claude 3.5 and Gemini 2.5 showed a systematic tendency toward overestimation, resulting in inflated patient-level symptom burden, whereas GPT-4o displayed a more balanced error profile without such inflation. Expert adjudication indicated that a substantial proportion of disagreements stemmed from the ambiguity inherent in the transcripts, whereas clear misclassifications were dominated by the misapplication of clinical thresholds. Exploratory meta-evaluation suggested that lower mapping and higher structure or expansion scores may accompany larger rating errors.</p><p>Evidence of LLM performance in clinical medicine has been largely generated from clinician-authored records or other proxy texts [<xref ref-type="bibr" rid="ref9">9</xref>]. Although a recent study indicated that LLMs can detect depression and anxiety in patients with chronic disease contexts [<xref ref-type="bibr" rid="ref22">22</xref>], relatively few studies have evaluated models using semistructured psychiatric interviews, often in narrower settings [<xref ref-type="bibr" rid="ref17">17</xref>]. The present exploratory findings extend existing evidence by showing high aggregate concordance across all 3 evaluated models for a heterogeneous set of 39 symptoms and interference items using real-world psycho-oncology interviews conducted in a non-English clinical setting. By directly evaluating original Korean interview transcripts rather than translated or proxy texts, this study found high aggregate concordance between the 3 evaluated LLMs and psycho-oncologists&#x2019; item-level ratings in this Korean-language psycho-oncology sample. While these findings should be interpreted as exploratory evidence of model behavior rather than as definitive validation for clinical deployment, prospective, workflow-specific studies may examine whether transcript-derived item-level ratings could serve as supplementary information for clinician-supervised postinterview review, with clinicians retaining responsibility for interpretation and clinical decisions. The findings also contribute to addressing gaps associated with the English-dominant and Western-centric nature of contemporary LLM training and benchmarking [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. Moreover, the clinician-adjudicated error review and structured meta-evaluation of model rationales provided complementary exploratory analyses of how models diverged from clinician judgment. These analyses suggest that models with similar aggregate agreement may still differ in directional bias, recurrent error patterns, and explanatory behavior, supporting the need to examine model performance beyond single-number summary metrics [<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>This study focused on item-level symptom ratings to better align with the requirements of granular symptom monitoring in oncologic care [<xref ref-type="bibr" rid="ref24">24</xref>]. From a model-evaluation perspective, decomposing assessment into individual items offers a more transparent and reviewable approach than asking a model to assign a diagnosis directly. Item-level ratings enable disagreements to be localized and may, in principle, be combined using prespecified external rules, consistent with structured scoresheet approaches [<xref ref-type="bibr" rid="ref25">25</xref>]. This modular framework supports evaluating item-level agreement as a distinct component of LLM-based psychiatric assessment and motivates future research examining whether such ratings can contribute to reliable syndrome-level classification.</p><p>Although the 3 LLMs exhibited broadly similar overall agreement with the psycho-oncologists, they differed in the distribution and direction of errors. Previous LLM-based depression-screening research has likewise reported variation in performance across models and symptom items [<xref ref-type="bibr" rid="ref26">26</xref>]. In our study, Claude 3.5 had the highest sensitivity and NPV point estimates but also generated more false-positive classifications, while Claude 3.5 and Gemini 2.5 yielded higher patient-level symptom counts than the clinician ratings. In contrast, GPT-4o had the highest PPV and did not show a statistically significant increase in patient-level symptom counts. These findings suggest that aggregate agreement alone may obscure differences in the balance between missed and overidentified symptoms. If replicated across model versions and independent samples, such differences may help clarify the relevance of particular error profiles across assessment contexts.</p><p>In this study, clinician adjudication helped distinguish definite model errors from disagreements arising from incomplete, indirect, or context-dependent transcript evidence. Such ambiguity may reflect a characteristic of semistructured psychiatric interviews and may limit the extent to which definitive judgments can be made from interview transcripts alone [<xref ref-type="bibr" rid="ref11">11</xref>]. Within the subset of disagreements judged to represent definite misclassification, the most common mechanism across the models was the misapplication of severity thresholds. This pattern is potentially important because threshold errors shift the decision boundary and may preferentially increase false positives or false negatives. Other errors were less frequent but particularly salient in psycho-oncology, where treatment-related neurovegetative symptoms closely mimic affective distress [<xref ref-type="bibr" rid="ref1">1</xref>]. Consistent with prior studies, the tendency to interpret somatic symptoms as depressive indicators represents a common challenge in AI-based clinical reasoning [<xref ref-type="bibr" rid="ref27">27</xref>]. The clinician-adjudicated taxonomy provides a structured account of recurrent disagreement mechanisms and may help identify priorities for model refinement. The predominance of threshold errors suggests that future prompt-development studies should examine whether clearer operationalization of severity cutoffs and stronger anchoring to the target time window improve rating consistency. Similarly, the occurrence of somatic confounding highlights the importance of evaluating strategies that distinguish cancer- or treatment-related neurovegetative symptoms from psychological distress. In this way, the taxonomy offers an empirically grounded framework for developing and testing targeted error-mitigation strategies, consistent with prior work on prompt-based mitigation in medical LLMs [<xref ref-type="bibr" rid="ref28">28</xref>].</p><p>Rationale characteristics also differed across models. Claude 3.5 produced more structured and citation-dense justifications, whereas GPT-4o was lower on these dimensions and Gemini 2.5 generally showed an intermediate profile. Exploratory mixed-effects analyses suggested that more structured or evidence-rich rationales did not necessarily correspond to closer agreement with psychiatrist ratings. This observation is broadly consistent with prior studies showing that eliciting explicit reasoning may improve interpretability without reliably improving task performance [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. The analyses also suggested that weaker mapping and greater interpretive expansion may accompany larger absolute error across source models. These observations support further examination of prompting strategies that more explicitly connect transcript evidence with scoring criteria while discouraging unsupported extrapolation. However, greater expansion may also be a marker of difficult or ambiguous cases that elicit both greater interpretive extrapolation and greater model-clinician disagreement, rather than a cause of error. Because the present study evaluated observable rationale characteristics rather than the models&#x2019; internal decision processes, it does not establish whether the generated rationales reflected the processes underlying the final outputs [<xref ref-type="bibr" rid="ref31">31</xref>].</p><p>Several limitations should be noted. First, this exploratory study included 101 patients from Korean-language psycho-oncology settings and predominantly represented women and patients with breast cancer, with limited representation of survivorship settings. The 3931 item-level ratings were clustered within these 101 interviews and did not constitute independent patient-level observations, limiting subgroup analyses. Accordingly, the generalizability of the findings to male patients, other cancer types, survivorship populations, non-Korean linguistic and cultural contexts, or nononcologic psychiatric populations should be examined in larger, more diverse, and independent cohorts. Second, although interrater reliability among psychiatrists was high, the substantial proportion of ambiguous cases highlights that some degree of interpretive variability is intrinsic to semistructured psychiatric interviews. Third, the analyses were based on text transcripts, whereas clinicians had access to nonverbal cues integral to mental status assessment. Fourth, the use of time-stamped clinician labels enables item-specific segmentation, which may represent a more favorable setting than typical transcript- or STT-based workflows. Fifth, case-level adjudication of the full mismatch set was performed by 1 primary psychiatrist. While a second independent reviewer showed high agreement for the ambiguous-versus-definite distinction in a stratified audit of 150 cases (Cohen &#x03BA;=0.82), residual subjectivity remains possible. Moreover, although no broad clinical-domain or patient subgroup heterogeneity was detected, ambiguous adjudications constituted a substantial proportion of mismatches and differed between GPT-4o and Claude 3.5. Sixth, in the 10-patient subset, agreement between the mean automated meta-evaluation scores and human ratings ranged from ICC (A,1)=0.646 to 0.822 across dimensions, with moderate reliability for citation and structure. This residual measurement uncertainty and the possibility of model-family-specific bias despite the multievaluator approach represent important limitations, and the findings based on the automated meta-evaluation scores should therefore be interpreted as exploratory. Finally, the evaluation was limited to specific model versions available at the time of analysis; future updates or fine-tuned variants may exhibit different error profiles.</p><p>In conclusion, this exploratory clinician-benchmarked evaluation showed that contemporary LLMs could approximate symptom-level assessment in a Korean psycho-oncology sample composed predominantly of women and patients with breast cancer. Despite high aggregate concordance, the models differed in directional bias and disagreement patterns, suggesting that similar overall performance may obscure potentially meaningful differences in their observed error profiles. These findings require further validation in larger, more diverse, and independent cohorts.</p></sec></body><back><ack><p>We would like to thank Editage [<xref ref-type="bibr" rid="ref32">32</xref>] for English language editing.</p><p>The ChatGPT models GPT-5.4, GPT-5.5, and GPT-5.6 Sol (OpenAI) were used under full human supervision to generate, proofread, and edit the text; translate selected passages; reformat selected materials; and assist with quality assessment. The use of large language models as study tools is described separately in the Methods section. All AI-assisted outputs were reviewed and approved by the authors.</p></ack><notes><sec><title>Funding</title><p>This research was supported by a grant from the Korea Health Technology R&#x0026;D Project through the Korea Health Industry Development Institute (KHIDI), funded by the Ministry of Health and Welfare, Republic of Korea (grant number: RS-2023-KH134997). The funder had no role in the study design, data collection, analysis, interpretation of the results, manuscript preparation, or decision to submit the manuscript.</p></sec><sec><title>Data Availability</title><p>The data analyzed in this study were collected with the approval of the institutional review board and contained highly sensitive personal and medical information. These data were made available to the present study for research purposes only and are not publicly available because of ethical restrictions, Korean privacy regulations, and institutional policies governing the protection of identifiable mental health data. Access to the underlying annotated interview materials may be considered upon reasonable request, subject to approval by the relevant institutional review boards and data governance committees, and solely for audit or verification purposes. Any such request must comply with the applicable ethical and legal requirements and should be directed to the corresponding author.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: DY, BJH</p><p>Data curation: HY, DY</p><p>Formal analysis: JK, HY</p><p>Funding acquisition: BJH</p><p>Investigation: JK, KLS, CWY, WHK, HY</p><p>Methodology: JK, KLS, CWY, WHK, SHL, JSS, HY, DY, BJH</p><p>Software: HY, DY</p><p>Supervision: BJH</p><p>Writing &#x2013; original draft: JK</p><p>Writing &#x2013; review and editing: KLS, CWY, WHK, SHL, JSS, DY, BJH</p><p>All authors read and approved the final manuscript</p></fn><fn fn-type="conflict"><p>DY is a co-founder of Genesis Lab Inc. HY is an employee of Genesis Lab Inc. BJH serves as a nonexecutive director of Genesis Lab Inc. Genesis Lab Inc. provided no financial or material support for this study. DY, HY, and BJH declare no other financial or nonfinancial competing interests. All other authors declare no conflicts of interest.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">GPT</term><def><p>generative pretrained transformer</p></def></def-item><def-item><term id="abb2">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb3">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb4">NPV</term><def><p>negative predictive value</p></def></def-item><def-item><term id="abb5">PPV</term><def><p>positive predictive value</p></def></def-item><def-item><term id="abb6">QWK</term><def><p>quadratic weighted kappa</p></def></def-item><def-item><term id="abb7">STT</term><def><p>speech-to-text</p></def></def-item><def-item><term id="abb8">TRIPOD-LLM</term><def><p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis&#x2013;Large Language Model</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mitchell</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bhatti</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Prevalence of depression, anxiety, and adjustment disorder in oncological, haematological, and palliative-care settings: a meta-analysis of 94 interview-based studies</article-title><source>Lancet Oncol</source><year>2011</year><month>02</month><volume>12</volume><issue>2</issue><fpage>160</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.1016/S1470-2045(11)70002-X</pub-id><pub-id pub-id-type="medline">21251875</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rodin</surname><given-names>G</given-names> </name><name name-style="western"><surname>Feldman</surname><given-names>A</given-names> </name><name name-style="western"><surname>Trapani</surname><given-names>D</given-names> </name><etal/></person-group><article-title>The human crisis in cancer: a Lancet Oncology Commission</article-title><source>Lancet Oncol</source><year>2025</year><month>12</month><volume>26</volume><issue>12</issue><fpage>e628</fpage><lpage>e670</lpage><pub-id pub-id-type="doi">10.1016/S1470-2045(25)00530-3</pub-id><pub-id pub-id-type="medline">41192457</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>DiMatteo</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Lepper</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Croghan</surname><given-names>TW</given-names> </name></person-group><article-title>Depression is a risk factor for noncompliance with medical treatment: meta-analysis of the effects of anxiety and depression on patient adherence</article-title><source>Arch Intern Med</source><year>2000</year><month>07</month><day>24</day><volume>160</volume><issue>14</issue><fpage>2101</fpage><lpage>2107</lpage><pub-id pub-id-type="doi">10.1001/archinte.160.14.2101</pub-id><pub-id pub-id-type="medline">10904452</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nederhof</surname><given-names>AJ</given-names> </name></person-group><article-title>Methods of coping with social desirability bias: a review</article-title><source>Euro J Social Psych</source><year>1985</year><month>07</month><volume>15</volume><issue>3</issue><fpage>263</fpage><lpage>280</lpage><pub-id pub-id-type="doi">10.1002/ejsp.2420150303</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>First</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Williams</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Karg</surname><given-names>RS</given-names> </name><name name-style="western"><surname>Spitzer</surname><given-names>RL</given-names> </name></person-group><article-title>The structured clinical interview for DSM-5&#x00AE;</article-title><source>American Psychiatric Association Publishing</source><year>2015</year><access-date>2026-08-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.appi.org/products/structured-clinical-interview-for-dsm-5-scid-5">https://www.appi.org/products/structured-clinical-interview-for-dsm-5-scid-5</ext-link></comment></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>The Lancet Digital Health</collab></person-group><article-title>Large language models: a new chapter in digital health</article-title><source>Lancet Digit Health</source><year>2024</year><month>01</month><volume>6</volume><issue>1</issue><fpage>e1</fpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00254-6</pub-id><pub-id pub-id-type="medline">38123249</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Clay</surname><given-names>B</given-names> </name><name name-style="western"><surname>Bergman</surname><given-names>HI</given-names> </name><name name-style="western"><surname>Salim</surname><given-names>S</given-names> </name><name name-style="western"><surname>Pergola</surname><given-names>G</given-names> </name><name name-style="western"><surname>Shalhoub</surname><given-names>J</given-names> </name><name name-style="western"><surname>Davies</surname><given-names>AH</given-names> </name></person-group><article-title>Natural language processing techniques applied to the electronic health record in clinical research and practice&#x2014;an introduction to methodologies</article-title><source>Comput Biol Med</source><year>2025</year><month>04</month><volume>188</volume><fpage>109808</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2025.109808</pub-id><pub-id pub-id-type="medline">39946783</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hossain</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rana</surname><given-names>R</given-names> </name><name name-style="western"><surname>Higgins</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Natural language processing in electronic health records in relation to healthcare decision-making: a systematic review</article-title><source>Comput Biol Med</source><year>2023</year><month>03</month><volume>155</volume><fpage>106649</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106649</pub-id><pub-id pub-id-type="medline">36805219</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bedi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Orr-Ewing</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Testing and evaluation of health care applications of large language models: a systematic review</article-title><source>JAMA</source><year>2025</year><month>01</month><day>28</day><volume>333</volume><issue>4</issue><fpage>319</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1001/jama.2024.21700</pub-id><pub-id pub-id-type="medline">39405325</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koutsouleris</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hauser</surname><given-names>TU</given-names> </name><name name-style="western"><surname>Skvortsova</surname><given-names>V</given-names> </name><name name-style="western"><surname>De Choudhury</surname><given-names>M</given-names> </name></person-group><article-title>From promise to practice: towards the realisation of AI-informed mental health care</article-title><source>Lancet Digit Health</source><year>2022</year><month>11</month><volume>4</volume><issue>11</issue><fpage>e829</fpage><lpage>e840</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(22)00153-4</pub-id><pub-id pub-id-type="medline">36229346</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Andreasen</surname><given-names>NC</given-names> </name></person-group><article-title>DSM and the death of phenomenology in America: an example of unintended consequences</article-title><source>Schizophr Bull</source><year>2007</year><month>01</month><volume>33</volume><issue>1</issue><fpage>108</fpage><lpage>112</lpage><pub-id pub-id-type="doi">10.1093/schbul/sbl054</pub-id><pub-id pub-id-type="medline">17158191</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pennebaker</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Mehl</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Niederhoffer</surname><given-names>KG</given-names> </name></person-group><article-title>Psychological aspects of natural language use: our words, our selves</article-title><source>Annu Rev Psychol</source><year>2003</year><volume>54</volume><issue>1</issue><fpage>547</fpage><lpage>577</lpage><pub-id pub-id-type="doi">10.1146/annurev.psych.54.101601.145041</pub-id><pub-id pub-id-type="medline">12185209</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>EE</given-names> </name><name name-style="western"><surname>Torous</surname><given-names>J</given-names> </name><name name-style="western"><surname>De Choudhury</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Artificial intelligence for mental health care: clinical applications, barriers, facilitators, and artificial wisdom</article-title><source>Biol Psychiatry Cogn Neurosci Neuroimaging</source><year>2021</year><month>09</month><volume>6</volume><issue>9</issue><fpage>856</fpage><lpage>864</lpage><pub-id pub-id-type="doi">10.1016/j.bpsc.2021.02.001</pub-id><pub-id pub-id-type="medline">33571718</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dodge</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sap</surname><given-names>M</given-names> </name><name name-style="western"><surname>Marasovi&#x0107;</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Documenting large webtext corpora: a case study on the Colossal Clean Crawled Corpus</article-title><source>Proc 2021 Conf Empir Methods Nat Lang Process</source><year>2021</year><fpage>1286</fpage><lpage>1305</lpage><pub-id pub-id-type="doi">10.18653/v1/2021.emnlp-main.98</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Hwang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Roh</surname><given-names>HW</given-names> </name><name name-style="western"><surname>Park</surname><given-names>RW</given-names> </name></person-group><article-title>Performance of open-source large language models in psychiatry: usability study through comparative analysis of non-English records and English translations</article-title><source>J Med Internet Res</source><year>2025</year><month>08</month><day>18</day><volume>27</volume><fpage>e69857</fpage><pub-id pub-id-type="doi">10.2196/69857</pub-id><pub-id pub-id-type="medline">40825309</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hoffmann</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rueger</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Current applications and challenges in large language models for patient care: a systematic review</article-title><source>Commun Med (Lond)</source><year>2025</year><month>01</month><day>21</day><volume>5</volume><issue>1</issue><fpage>26</fpage><pub-id pub-id-type="doi">10.1038/s43856-024-00717-2</pub-id><pub-id pub-id-type="medline">39838160</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Thygesen</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Farrington</surname><given-names>J</given-names> </name><name name-style="western"><surname>Keen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name></person-group><article-title>Large language models for mental health applications: systematic review</article-title><source>JMIR Ment Health</source><year>2024</year><month>10</month><day>18</day><volume>11</volume><fpage>e57400</fpage><pub-id pub-id-type="doi">10.2196/57400</pub-id><pub-id pub-id-type="medline">39423368</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gascon</surname><given-names>B</given-names> </name><name name-style="western"><surname>Elman</surname><given-names>J</given-names> </name><name name-style="western"><surname>Macedo</surname><given-names>A</given-names> </name><name name-style="western"><surname>Leung</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Rodin</surname><given-names>G</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name></person-group><article-title>Two-step screening for depression and anxiety in patients with cancer: a retrospective validation study using real-world data</article-title><source>Curr Oncol</source><year>2024</year><month>10</month><day>23</day><volume>31</volume><issue>11</issue><fpage>6488</fpage><lpage>6501</lpage><pub-id pub-id-type="doi">10.3390/curroncol31110481</pub-id><pub-id pub-id-type="medline">39590112</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hua</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Na</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A scoping review of large language models for generative tasks in mental health care</article-title><source>NPJ Digit Med</source><year>2025</year><month>04</month><day>30</day><volume>8</volume><issue>1</issue><fpage>230</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01611-4</pub-id><pub-id pub-id-type="medline">40307331</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Landis</surname><given-names>JR</given-names> </name><name name-style="western"><surname>King</surname><given-names>TS</given-names> </name><name name-style="western"><surname>Choi</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Chinchilli</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Koch</surname><given-names>GG</given-names> </name></person-group><article-title>Measures of agreement and concordance with clinical research applications</article-title><source>Stat Biopharm Res</source><year>2011</year><month>05</month><volume>3</volume><issue>2</issue><fpage>185</fpage><lpage>209</lpage><pub-id pub-id-type="doi">10.1198/sbr.2011.10019</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gallifant</surname><given-names>J</given-names> </name><name name-style="western"><surname>Afshar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ameen</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The TRIPOD-LLM reporting guideline for studies using large language models</article-title><source>Nat Med</source><year>2025</year><month>01</month><volume>31</volume><issue>1</issue><fpage>60</fpage><lpage>69</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03425-5</pub-id><pub-id pub-id-type="medline">39779929</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>SP</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>ML</given-names> </name><etal/></person-group><article-title>Optimizing large language models for detecting symptoms of depression/anxiety in chronic diseases patient communications</article-title><source>NPJ Digit Med</source><year>2025</year><month>09</month><day>30</day><volume>8</volume><issue>1</issue><fpage>580</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01969-5</pub-id><pub-id pub-id-type="medline">41028413</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hua</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Large language models in mental health care: a scoping review</article-title><source>Curr Treat Options Psych</source><year>2025</year><volume>12</volume><issue>1</issue><pub-id pub-id-type="doi">10.1007/s40501-025-00363-y</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Basch</surname><given-names>E</given-names> </name><name name-style="western"><surname>Reeve</surname><given-names>BB</given-names> </name><name name-style="western"><surname>Mitchell</surname><given-names>SA</given-names> </name><etal/></person-group><article-title>Development of the National Cancer Institute&#x2019;s patient-reported outcomes version of the common terminology criteria for adverse events (PRO-CTCAE)</article-title><source>J Natl Cancer Inst</source><year>2014</year><month>09</month><volume>106</volume><issue>9</issue><fpage>dju244</fpage><pub-id pub-id-type="doi">10.1093/jnci/dju244</pub-id><pub-id pub-id-type="medline">25265940</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>K</given-names> </name><name name-style="western"><surname>Rasool</surname><given-names>A</given-names> </name><name name-style="western"><surname>Surabhi</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Aiding large language models using clinical scoresheets for neurobehavioral diagnostic classification from text: algorithm development and validation</article-title><source>JMIR AI</source><year>2025</year><month>10</month><day>21</day><volume>4</volume><fpage>e75030</fpage><pub-id pub-id-type="doi">10.2196/75030</pub-id><pub-id pub-id-type="medline">41118647</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Teferra</surname><given-names>BG</given-names> </name><name name-style="western"><surname>Perivolaris</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hsiang</surname><given-names>WN</given-names> </name><etal/></person-group><article-title>Leveraging large language models for automated depression screening</article-title><source>PLOS Digit Health</source><year>2025</year><month>07</month><volume>4</volume><issue>7</issue><fpage>e0000943</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000943</pub-id><pub-id pub-id-type="medline">40720397</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abbasian</surname><given-names>M</given-names> </name><name name-style="western"><surname>Khatibi</surname><given-names>E</given-names> </name><name name-style="western"><surname>Azimi</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Foundation metrics for evaluating effectiveness of healthcare conversations powered by generative AI</article-title><source>NPJ Digit Med</source><year>2024</year><month>03</month><day>29</day><volume>7</volume><issue>1</issue><fpage>82</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01074-z</pub-id><pub-id pub-id-type="medline">38553625</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmidgall</surname><given-names>S</given-names> </name><name name-style="western"><surname>Harris</surname><given-names>C</given-names> </name><name name-style="western"><surname>Essien</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Evaluation and mitigation of cognitive biases in medical language models</article-title><source>NPJ Digit Med</source><year>2024</year><month>10</month><day>21</day><volume>7</volume><issue>1</issue><fpage>295</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01283-6</pub-id><pub-id pub-id-type="medline">39433945</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gu</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kr&#x00FC;ger</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>J</given-names> </name></person-group><article-title>Why chain of thought fails in clinical text understanding</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 26, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2509.21933</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Savage</surname><given-names>T</given-names> </name><name name-style="western"><surname>Nayak</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>R</given-names> </name><name name-style="western"><surname>Rangan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>JH</given-names> </name></person-group><article-title>Diagnostic reasoning prompts reveal the potential for large language model interpretability in medicine</article-title><source>NPJ Digit Med</source><year>2024</year><month>01</month><day>24</day><volume>7</volume><issue>1</issue><fpage>20</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01010-1</pub-id><pub-id pub-id-type="medline">38267608</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jacovi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Goldberg</surname><given-names>Y</given-names> </name></person-group><article-title>Towards faithfully interpretable NLP systems: how should we define and evaluate faithfulness?</article-title><source>Proc 58th Annu Meet Assoc Comput Linguist</source><year>2020</year><fpage>4198</fpage><lpage>4205</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.386</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="web"><source>Editage [Article in Korean]</source><access-date>2026-08-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.editage.co.kr/">https://www.editage.co.kr/</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Zero-shot prompt for large language model&#x2013;based symptom evaluation, reliability analyses of automated meta-evaluation scores against human reference ratings, and meta-evaluation prompts and scoring rubrics.</p><media xlink:href="jmir_v28i1e91405_app1.docx" xlink:title="DOCX File, 38 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Deidentified examples of clinician&#x2013;large language model disagreements adjudicated as ambiguous.</p><media xlink:href="jmir_v28i1e91405_app2.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Ambiguity-excluded binary screening performance, ambiguous-disagreement distributions and regression analyses, and associations between meta-evaluation scores and absolute rating error.</p><media xlink:href="jmir_v28i1e91405_app3.docx" xlink:title="DOCX File, 24 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Ordinal confusion matrices comparing model-generated and clinician symptom-severity ratings and domain-adjusted mixed-effects estimates of associations between automated and human-reference meta-evaluation scores and absolute rating error.</p><media xlink:href="jmir_v28i1e91405_app4.docx" xlink:title="DOCX File, 184 KB"/></supplementary-material><supplementary-material id="app5"><label>Checklist 1</label><p>TRIPOD-LLM checklist.</p><media xlink:href="jmir_v28i1e91405_app5.pdf" xlink:title="PDF File, 1356 KB"/></supplementary-material></app-group></back></article>