<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e92374</article-id><article-id pub-id-type="doi">10.2196/92374</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Finite State Machine&#x2013;Guided Retrieval-Augmented Generation Improves Expert-Rated Acceptability of a Peripherally Inserted Central Catheter Self-Management Chatbot: Single-Center Content Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Lee</surname><given-names>Mangyeong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Cho</surname><given-names>Seung-Beom</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yu</surname><given-names>Jae-Wook</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yoon</surname><given-names>Junghee</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Choi</surname><given-names>Jae-Boong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jung</surname><given-names>Kyu-Hwan</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Cho</surname><given-names>Juhee</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff7">7</xref></contrib></contrib-group><aff id="aff1"><institution>Center for Clinical Epidemiology, Samsung Medical Center</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff2"><institution>Department of Digital Health, Samsung Advanced Institute for Health Sciences and Technology, Sungkyunkwan University</institution><addr-line>115, Irwon-ro</addr-line><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff3"><institution>School of Mechanical Engineering, Sungkyunkwan University</institution><addr-line>Suwon</addr-line><country>Republic of Korea</country></aff><aff id="aff4"><institution>Department of Clinical Research Design and Evaluation, Samsung Advanced Institute for Health Sciences and Technology, Sungkyunkwan University</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff5"><institution>Department of Medical Device Management and Research, Samsung Advanced Institute for Health Sciences and Technology, Sungkyunkwan University</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff6"><institution>Clinical Robotics and Embodied AI Research Center, Smart Healthcare Research Institute, Research Institute for Future Medicine, Samsung Medical Center</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff7"><institution>Cancer Education Center, Samsung Medical Center</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Shin</surname><given-names>Donghoon</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Chatzimina</surname><given-names>Maria</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Juhee Cho, PhD, Department of Digital Health, Samsung Advanced Institute for Health Sciences and Technology, Sungkyunkwan University, 115, Irwon-ro, Seoul, Republic of Korea, 82 234101448; <email>jcho@skku.edu</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>11</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e92374</elocation-id><history><date date-type="received"><day>28</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>30</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>30</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Mangyeong Lee, Seung-Beom Cho, Jae-Wook Yu, Junghee Yoon, Jae-Boong Choi, Kyu-Hwan Jung, Juhee Cho. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 11.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e92374"/><abstract><sec><title>Background</title><p>Patients with cancer undergoing long-term or vesicant chemotherapy frequently require peripherally inserted central catheters (PICCs). Due to the nature of ambulatory treatment administration, self-PICC management is essential for the continuation and completion of the planned treatment. Large language models offer potential for continuous patient support, but hallucinations and insufficient adherence to clinical protocols remain concerns. Fine-tuning (FT) and retrieval-augmented generation (RAG) improve factual grounding but cannot enforce the structured decision logic of expert-led PICC consultations, leaving the value of dialogue-control mechanisms unclear.</p></sec><sec><title>Objective</title><p>This study is an expert-panel evaluation designed to exploratorily verify whether a finite state machine (FSM)&#x2013;guided RAG architecture yields incremental gains in clinical acceptability when added on top of FT and RAG compared with simpler architectures (fine-tuned model alone; fine-tuned model with RAG).</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a blinded ablation comparison of three chatbot architectures sharing an identical fine-tuned GPT-4o-mini base: model 1 (fine-tuning only; FT), model 2 (FT + RAG), and model 3 (FT + RAG + FSM). Three nurses specializing in PICC management (&#x2265;10 y experience) independently evaluated responses to 43 PICC-related queries. The two-stage evaluation comprised query-level forced-choice preference and global Likert ratings across 8 predefined dimensions. Interrater agreement was quantified with Gwet AC1. Preference rates were compared with Cochran Q and pairwise McNemar tests with Bonferroni correction. Global Likert ratings were summarized descriptively across 8 predefined dimensions.</p></sec><sec sec-type="results"><title>Results</title><p>The FSM-guided model (model 3) was preferred in 34 of 43 scenarios (79.1%). Interrater agreement was moderate (Gwet AC1=0.58, 95% CI 0.42&#x2010;0.75; <italic>P</italic>&#x003C;.001). Differences in model preference were significant (Cochran Q=48.2; <italic>P</italic>&#x003C;.001). Pairwise McNemar tests showed model 3 was preferred significantly more often than model 1 and model 2 (both <italic>P</italic>&#x003C;.001), while model 1 versus model 2 did not differ after Bonferroni correction. For the global ratings, descriptive statistics indicated that model 3 received higher ratings than models 1 and 2 across most criteria, although it tended to receive lower ratings for efficiency, possibly reflecting its multiturn structure. In qualitative debriefing, the nurses noted occasional unnecessary conversational turns under FSM guidance.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In this expert-panel content-validation study, the FSM-guided fine-tuned RAG model received higher expert-rated acceptability than the simpler architectures, with FSM-based dialogue control more closely aligning chatbot outputs with expert-led PICC consultation patterns. Patient-facing usability testing and multicenter validation are required before claims regarding patient safety or clinical deployment can be made.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>retrieval-augmented generation</kwd><kwd>finite state machine</kwd><kwd>conversational agent</kwd><kwd>peripherally inserted central catheter</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Patients undergoing chemotherapy frequently require peripherally inserted central catheters (PICC) for the administration of vesicant agents or prolonged treatment regimens [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. However, long-term PICC insertion carries a significant risk of complications, including infection and thrombosis. One study reported that approximately half (48.1%) of patients with cancer experienced at least one complication during PICC use [<xref ref-type="bibr" rid="ref3">3</xref>]. In ambulatory settings, patient competency in catheter management is particularly associated with complications, as evidenced by a decrease in complication rates from 40% to 15.9% following the implementation of PICC management education [<xref ref-type="bibr" rid="ref4">4</xref>]. While systematic education and counseling for PICC management are crucial, self-care procedures, including disinfection, dressing changes, and heparin flushing, present complex and demanding technical challenges for patients [<xref ref-type="bibr" rid="ref5">5</xref>]. Consequently, timely assistance is critical when patients encounter concerns or difficulties in catheter management during daily activities [<xref ref-type="bibr" rid="ref6">6</xref>]. However, patients with PICCs typically face limited access to continuous health care provider communication and subsequently lack adequate support systems to address unexpected situations [<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>Medical chatbot technology has emerged as a promising solution for resolving these unmet needs. Chatbots offer 24-hour availability with immediate response capabilities, enabling prompt assistance when patients experience urgent concerns or discomfort, while providing services independent of physical distance [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. When empathizing with patient-facing concerns and providing personalized advice, chatbots can deliver emotional support, such as anxiety reduction [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. In particular, incorporating large language models (LLMs) can provide superior information quality and user experience compared with conventional instruction delivery [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>Despite these potential benefits, significant concerns and challenges remain regarding the clinical implementation of LLM-based chatbots. The most critical issue is the hallucinations of LLMs. This error manifests as a plausible generation of unfounded content inconsistent with the given questions or contexts, creating risks that general users or patients may misinterpret as factual information [<xref ref-type="bibr" rid="ref14">14</xref>]. Thus, there is a prevailing perception that rule-based systems remain safer in clinical settings, even within limited scenarios [<xref ref-type="bibr" rid="ref15">15</xref>]. To address these technological limitations, fine-tuning (FT) and Retrieval-Augmented Generation (RAG) techniques have been considered [<xref ref-type="bibr" rid="ref16">16</xref>-<xref ref-type="bibr" rid="ref18">18</xref>]. FT involves additional training with medical domain data to internalize specialized knowledge, whereas RAG produces evidence-based responses through real-time reference to external trustworthy knowledge bases [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>].</p><p>Although FT and RAG can improve factual grounding, they have inherent limitations in structured clinical consultations [<xref ref-type="bibr" rid="ref18">18</xref>]. Fine-tuned LLMs approximate clinical logic probabilistically and may deviate from the required procedural steps, whereas RAG constrains content but cannot enforce the mandatory sequence or branching rules of guideline-based consultations. In PICC self-management, where symptom triage, safety checks, and decision branching must follow predefined clinical workflows, these limitations result in insufficient procedural validity.</p><p>To address this gap, we introduced a finite-state machine (FSM) within a fine-tuned RAG system. An FSM represents consultation as a set of discrete states, such as symptom assessment, severity evaluation, and escalation decisions, with allowed transitions strictly determined by clinical protocols. Constraining the LLM to operate only within verified state transitions preserves the determinism and stability of rule-based systems while retaining the linguistic flexibility of LLMs. Consequently, an FSM-integrated RAG model may offer a more clinically grounded approach for structured consultations, such as PICC self-management. Therefore, this study is a proof-of-concept evaluation designed to exploratorily verify whether FSM-based dialogue control yields incremental gains in clinical acceptability when added on top of FT and RAG for an LLM-based PICC self-management chatbot, using expert content validation as the evaluation method.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Formalization of Consultation Protocol for PICC Management</title><p>To ensure comprehensive coverage of PICC-related issues, a knowledge base for a PICC chatbot was designed using educational materials employed at the Samsung Medical Center. These materials comprised general procedures for PICC insertion, potential symptoms or complications during catheter maintenance, and precautionary measures for maintenance, including dressing changes and heparin flushing. Additionally, daily activities prohibited or allowed to keep PICC safe were incorporated into the materials based on questions frequently asked by patients. Most content was similar to other hospitals&#x2019; educational materials and was based on patients&#x2019; supportive care needs from an empirical study [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref23">23</xref>].</p><p>To build structured consultation flows, two oncology-specialized nurses were invited to generate response scenarios using a binary decision (yes or no) tree format. For example, when a patient indicated fever symptoms, the guided dialogue was configured to elicit the current body temperature and assess whether it surpassed the threshold of 38&#x00B0;C. Subsequently, if the temperature exceeded the threshold, the manual directed the patients to seek immediate medical attention. All figures and tables within the materials were converted into structured text to enhance the efficiency with which the chatbot processed and retrieved relevant information.</p></sec><sec id="s2-2"><title>System Architecture Overview</title><p>A multilayered architecture integrating FT, RAG, and a FSM was developed to progressively strengthen protocol adherence in PICC self-management consultations, as illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>. In model 1, domain-specific responses are generated through FT. In model 2, responses are further grounded in guideline-based documents through RAG. In model 3, an FSM layer is introduced to constrain the dialogue flow according to predefined clinical decision pathways.</p><p>User queries are processed by the LLM to generate a JSON-based intent classification and confidence score. When the confidence exceeds a predefined threshold (&#x2265;0.8), the query is routed to the corresponding FSM; otherwise, the RAG pipeline is used as a fallback. Through this incremental design, each layer adds a distinct capability&#x2014;domain knowledge, evidence grounding, and deterministic control&#x2014;enabling structured and protocol-adherent medical dialogue.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Overall architecture of the peripherally inserted central catheter management chatbot across three experimental settings. Model 1 represents a fine-tuned GPT-4o-mini model trained on guideline-based dialogue data. Model 2 integrates the fine-tuned model with a retrieval-augmented generation system using vectorized guideline documents. Model 3 further extends this architecture by incorporating a finite-state machine for structured dialogue control, enabling contextually accurate and multiturn interactions. DB: database; FSM: finite state machine; FT: fine-tuned; LLM: large language model; PICC: peripherally inserted central catheter; RAG: retrieval-augmented generation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92374_fig01.png"/></fig></sec><sec id="s2-3"><title>Incremental Model Construction</title><p>To isolate the contribution of each architectural component, the three models were constructed incrementally such that each subsequent model retained all elements of the previous model and added a single functional layer.</p><sec id="s2-3-1"><title>Model 1: Knowledge Internalization Layer</title><p>In model 1, the GPT-4o-mini was fine-tuned to incorporate domain-specific knowledge relevant to PICC patient management. The training dataset was developed by analyzing frequently asked questions derived from PICC-specific educational materials, resulting in 50 dialogue sessions, totaling approximately 300 user-assistant message pairs. Each session consisted of an average of 5&#x2010;6 multiturn exchanges to reflect realistic consultation scenarios. FT was conducted using the OpenAI platform, through which the model was optimized to generate responses aligned with the specific domains of PICC care and patient management.</p></sec><sec id="s2-3-2"><title>Model 2: Addition of a RAG-Based Context Search Layer</title><p>Model 2 was constructed by integrating a RAG system with fine-tuned model 1. The RAG mechanism enabled real-time searching of external knowledge bases, which were then incorporated into response generation. This approach provides access to detailed information and guideline updates not included in the FT dataset, thereby enhancing the knowledge coverage and response accuracy [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>].</p><p>The RAG pipeline was implemented using the &#x201D;LangChain<italic>&#x201D;</italic> (LangChain, Inc.) framework with the following configuration. First, the PICC educational materials (originally in PDF format) were extracted using &#x201C;PyPDF2&#x201D; (Mathieu Fenniak) and segmented into semantic units through paragraph-level chunking based on double newline delimiters. Each chunk was converted into vector representations using the OpenAI text-embedding-3-small model and stored in a Chroma vector database. Upon receiving a user query, the same embedding model was applied to vectorize the input, after which the top-k most relevant documents (k=3) were retrieved using cosine similarity-based search. The retrieved documents were incorporated into the prompt of the fine-tuned GPT-4o-mini model (OpenAI; temperature =0.1) using a &#x201C;stuff&#x201D; chain type via LangChain &#x201C;RetrievalQA&#x201D; chain. For multiturn conversations, &#x201C;ConversationalRetrievalChain&#x201D; from the LangChain was used, with up to five previous conversation pairs maintained for contextual continuity. A safety mechanism was implemented to provide a &#x201C;recommend a clinical visit&#x201D; message when no relevant documents were retrieved.</p></sec><sec id="s2-3-3"><title>Model 3: Addition of Deterministic Dialogue Control Layer Using FSM</title><p>Model 3 was constructed by integrating an FSM-based dialogue flow management system with model 2. The FSM layer encodes PICC-specific clinical workflows using predefined states and transition rules to ensure deterministic and guideline-concordant responses in safety-critical situations [<xref ref-type="bibr" rid="ref26">26</xref>]. A total of 30 FSMs were constructed based on the consultation protocol prespecified in the Samsung Medical Center, covering nine clinical domains including insertion-site abnormalities, catheter malfunction, symptom assessment, heparin management, and daily activity guidance, as summarized in <xref ref-type="table" rid="table1">Table 1</xref>. Each FSM encoded a structured consultation flow as a set of discrete states and binary (yes or no) conditional transitions, with 130 states and 72 conditional transitions identified across all FSMs. Conditional transitions represent decision points where the dialogue path diverges based on the patient&#x2019;s response. FSMs without conditional transitions (eg, FSM-5 and FSM-6) advance automatically through a fixed linear sequence of states, delivering state messages consecutively without branching.</p><p>Upon receiving a user query, the system performed intent classification by prompting the fine-tuned GPT-4o-mini model to return a JSON object containing the matched FSM identifier and a confidence score. Queries with confidence &#x2265;0.8 were routed to the corresponding FSM; those below this threshold were directed to the RAG pipeline as a fallback. Within an active FSM session, patient responses at each turn were classified as affirmative or negative through rule-based keyword matching for simple inputs and an LLM-based classification call (temperature =0) for ambiguous responses. If the LLM classification returned &#x201C;UNCLEAR,&#x201D; the active FSM session was terminated and the query was directly routed to the RAG fallback pipeline to ensure a timely response. As with all system outputs, the resulting responses were then converted into patient-friendly language through a separate LLM call.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Overview of the 30 finite state machines (FSM) for peripherally inserted central catheters (PICC) consultation.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">FSM<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> ID</td><td align="left" valign="bottom">Representative scenario</td><td align="left" valign="bottom">States</td><td align="left" valign="bottom">Conditional transitions</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Insertion site abnormalities</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-1</td><td align="left" valign="top">Bleeding at insertion site</td><td align="left" valign="top">7</td><td align="left" valign="top">4</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-2</td><td align="left" valign="top">Swelling at insertion site</td><td align="left" valign="top">4</td><td align="left" valign="top">2</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-3</td><td align="left" valign="top">Pus or discharge</td><td align="left" valign="top">4</td><td align="left" valign="top">2</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-4</td><td align="left" valign="top">Skin redness or allergic reaction</td><td align="left" valign="top">4</td><td align="left" valign="top">2</td></tr><tr><td align="left" valign="top" colspan="4">Catheter malfunction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-5</td><td align="left" valign="top">Catheter torn or punctured</td><td align="left" valign="top">3</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-6</td><td align="left" valign="top">Catheter dislodgement</td><td align="left" valign="top">5</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top" colspan="4">Symptom assessment</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-7</td><td align="left" valign="top">Arm swelling</td><td align="left" valign="top">4</td><td align="left" valign="top">3</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-8</td><td align="left" valign="top">Fever after heparin flush</td><td align="left" valign="top">4</td><td align="left" valign="top">3</td></tr><tr><td align="left" valign="top" colspan="4">Care scheduling</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><break/>FSM-9 ~ 11</td><td align="left" valign="top">Hospital visit timing, heparin schedule</td><td align="left" valign="top">2&#x2013;4</td><td align="left" valign="top">0&#x2013;2</td></tr><tr><td align="left" valign="top" colspan="4">Heparin management</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-12 ~ 15</td><td align="left" valign="top">Storage, dilution, obstruction, home pump</td><td align="left" valign="top">2&#x2013;3</td><td align="left" valign="top">1&#x2013;2</td></tr><tr><td align="left" valign="top" colspan="4">Equipment issues</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-16 ~ 18</td><td align="left" valign="top">Fixation device, supplies, hospital transfer</td><td align="left" valign="top">3&#x2013;5</td><td align="left" valign="top">1&#x2013;2</td></tr><tr><td align="left" valign="top" colspan="4">General PICC<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> care</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><break/>FSM-19</td><td align="left" valign="top">Duration of catheter use</td><td align="left" valign="top">2</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top" colspan="4">Daily activities</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-20 ~ 28</td><td align="left" valign="top">Exercise, bathing, driving, travel, arm use</td><td align="left" valign="top">2&#x2013;4</td><td align="left" valign="top">0&#x2013;2</td></tr><tr><td align="left" valign="top" colspan="4">Medical procedures</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FSM-29 ~ 30</td><td align="left" valign="top">Blood draw, blood pressure on PICC arm</td><td align="left" valign="top">2</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top" colspan="4">Total</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>30 FSMs</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">130</td><td align="left" valign="top">72</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>FSM: finite-state machine.</p></fn><fn id="table1fn2"><p><sup>b</sup>PICC: peripherally inserted central catheters.</p></fn><fn id="table1fn3"><p><sup>c</sup>Not applicable</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-3-4"><title>System Prompt Design</title><p>A standardized consultation system prompt was applied identically across all three models to control for prompt-related variation and isolate the effect of each architectural component. The prompt instructed the model to assume the role of an oncology-specialized nurse and contained seven directives: (1) listen to and empathize with the patient&#x2019;s concerns, (2) ask follow-up questions to accurately assess the situation, (3) determine whether the problem is PICC-related, (4) deliver clear warnings about dangerous behaviors, (5) provide guideline-based explanations in patient-friendly language, (6) answer strictly according to medical principles, and (7) assess whether clinical intervention is required. In model 3, two additional prompts were used for FSM-specific functions: an intent classification prompt for routing queries to the appropriate FSM via structured JSON output, and a response naturalization prompt for converting clinical state messages into conversational language. All prompts were written and deployed in Korean. The full text of all prompts, with English translations and API parameters, is provided in Section S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s2-4"><title>Expert-Rated Content Validity of LLM-Generated Responses to PICC-Related Queries</title><p>We employed an expert-panel content-validation approach, in which qualified clinical experts judged whether the LLM-generated responses met the standards of expert-led PICC consultation (<xref ref-type="fig" rid="figure2">Figure 2</xref>). This approach follows established methodology for clinical content validation, in which panels of three to five highly experienced experts judge whether an instrument or intervention meets domain-relevant clinical standards [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. It is also the modal practice in published LLM-in-health care evaluation, where median panel sizes of 2&#x2010;4 evaluators are typical [<xref ref-type="bibr" rid="ref29">29</xref>].</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Overview of the expert content validation procedure for LLM-generated responses on PICC self-management. In total, 43 queries were entered into three chatbot models (Models 1, 2, and 3), generating 129 blinded question&#x2013;response pairs. These were reviewed by three PICC-specialized nurse evaluators. Stage 1 assessed query-level model preference using a forced-choice procedure. Stage 2 elicited global ratings. After reviewing all 43 responses generated by a given model, each expert provided a single global rating on each of the 8 predefined evaluation dimensions using a 5-point Likert scale. A focus group interview was analyzed using inductive content analysis and triangulated with the quantitative findings. LLM: large language model; PICC: peripherally inserted central catheters.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92374_fig02.png"/></fig><p>The evaluation was conducted by a panel of three nurses specializing in infection control and PICC management, recruited via email between August and September 2025, all with at least 10 years of experience in PICC education, counseling, and clinical care for patients with cancer. A query set of 43 PICC-related questions was adopted from a previously published rule-based PICC chatbot whose questions were derived from real patient inquiries [<xref ref-type="bibr" rid="ref30">30</xref>]. The 43 queries were balanced across six PICC-management-related topics: (1) catheter care (n=10), (2) symptoms (n=4), (3) insertion site abnormalities (n=8), (4) heparin flushing (n=9), (5) daily activities (n=11), and (6) emergency situations (n=1). All identical queries were sequentially entered into the three models (models 1, 2, and 3), generating 129 responses (Section S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>The procedure consisted of two sequential stages: a query-level forced-choice preference evaluation, followed by a model-level global rating across eight predefined evaluation dimensions. For each of the 43 queries, each expert independently reviewed the three responses generated by models 1, 2, and 3 and selected the single response judged most appropriate as PICC self-management guidance. To minimize bias, no information regarding the underlying architecture of each model was disclosed to the experts, and all evaluations were conducted in a blinded manner based solely on the response texts. After completing the query-level evaluation for all 43 queries, each expert reviewed the full set of 43 responses generated by a given model and provided a single global rating for that model on each of eight predefined evaluation dimensions: accuracy, clarity, actionability, completeness, efficiency, adaptability, safety, and overall quality. These dimensions were defined based on prior studies of LLM-powered medical chatbots (<xref ref-type="other" rid="box1">Textbox 1</xref>) [<xref ref-type="bibr" rid="ref26">26</xref>-<xref ref-type="bibr" rid="ref31">31</xref>]. Each rating was provided on a 5-point Likert scale (1=highly inappropriate, 5=highly appropriate).</p><boxed-text id="box1"><title> Definition of the eight evaluation domains for global ratings.</title><list list-type="order"><list-item><p>Accuracy: Is the provided answer medically accurate and compliant with clinical principles in PICC management?</p></list-item><list-item><p>Clarity: Is the explanation sufficiently clear and straightforward for patients or caregivers to comprehend immediately?</p></list-item><list-item><p>Actionability: Does the response provide specific, executable instructions that patients can practically implement?</p></list-item><list-item><p>Completeness: Does the content of the generated response adequately address the patient&#x2019;s concern (or problem)?</p></list-item><list-item><p>Efficiency: Was core information delivered promptly without unnecessary conversational repetition?</p></list-item><list-item><p>Adaptability: Does the response consider diverse patient circumstances and provide situation-appropriate answers?</p></list-item><list-item><p>Safety: Does the response proactively prevent potential risk factors or patient misunderstandings?</p></list-item><list-item><p>Overall Quality: Comprehensive assessment of response quality</p></list-item></list></boxed-text></sec><sec id="s2-5"><title>Postevaluation Debriefing: Focus Group Interview</title><p>Following the quantitative performance evaluation of the three models, a qualitative focus group interview was conducted as a debriefing session to provide an in-depth interpretation of the evaluation results. The interview was conducted via the Zoom video conferencing platform (Eric Yuan<underline>)</underline> with the following objectives: (1) to thoroughly understand how each expert interpreted and applied the evaluation criteria, (2) reach a consensus on items where discordance existed among experts, and (3) identify specific areas for improvement, particularly for items that received low scores. The interview lasted for 1 hour and 10 minutes, and the entire session was audio-recorded with the participants&#x2019; consent.</p></sec><sec id="s2-6"><title>Statistical Analyses</title><p>Regarding expert preferences (forced-choice), the frequency of expert-selected models was aggregated to determine the distribution of preferred models per query, and the overall preference rate (%) for each model across all 43 queries was calculated. A winning model for each query was defined as one selected by at least two of the three evaluators. Interrater agreement was quantified with Gwet AC1, which provides more stable estimates than Fleiss&#x2019; &#x03BA; when preferences are imbalanced across categories [<xref ref-type="bibr" rid="ref31">31</xref>]. Differences in model-preference rates were tested with Cochran Q test, with post hoc pairwise comparisons using McNemar test with Bonferroni correction (corrected &#x03B1;=.0167).</p><p>Regarding the global ratings, descriptive statistics were used. All eight domains were presented as mean (&#x00B1; SD). For visual comparison of performance differences among the models, radar charts were constructed based on average scores across the eight evaluation dimensions. All analyses were performed using R (version 4.5.2; Posit PBC). Where inferential statistical tests were conducted, 2-sided tests were used, 95% CIs were reported when appropriate, and <italic>P</italic> values &#x003C;.05 were considered statistically significant unless adjusted for multiple comparisons using the Bonferroni method.</p><p>For the qualitative interview, the recorded material was first transcribed verbatim. Inductive content analysis was subsequently performed on the transcribed data to extract expert rationale and contextual examples that supplemented the quantitative findings. One researcher (ML) organized the coded content, and a second researcher (SB) independently confirmed the coding; discrepancies were resolved by consensus, with adjudication by a senior author (JY) where needed. Through this analysis, we identified the rationale underlying the quality of PICC consultation and the key elements for LLM-based chatbot improvement. The eight evaluation dimensions provided a loose initial framework for organizing codes, but additional themes were allowed to emerge from the narratives.</p></sec><sec id="s2-7"><title>Ethical Considerations</title><p>This study was reviewed and approved by the Institutional Review Board of Samsung Medical Center (Institutional Review Board number SMC 2025-07-146). All procedures were conducted in accordance with the Declaration of Helsinki. Informed consent was obtained from each evaluator by email prior to participation. Evaluator identities were replaced with anonymized codes (N01, N02, and N03) in all records, transcripts, and reported quotations. However, all evaluator-derived data were temporarily linked back to evaluators throughout the qualitative debriefing interviews. Each evaluator received an honorarium of 100,000 KRW (&#x2248;US $70) upon completion of the evaluation.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Comparative Evaluation of Information Quality Among Three Models</title><p>Based on the expert evaluations, model 3, an integrated architecture combining FT, RAG, and FSM, was most frequently selected to provide the most appropriate response across all topics in the PICC query set, followed by model 1 (<xref ref-type="fig" rid="figure3">Figure 3</xref>). Model 3 was the most frequently winning model for Catheter care (8/10, 80%), Symptoms (3/4, 75%), Insertion site abnormalities (7/8, 88%), Heparin flushing (6/9, 67%), Daily activities (9/11, 82%), and Emergency (1/1, 100%). However, interrater disagreement was observed in certain categories, particularly for Heparin flushing (1/9, 11%) and Daily activiti<italic>es</italic> (2/11, 18%).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Number of forced-choice expert preferences for each chatbot architecture across 6 categories of peripherally inserted central catheter (PICC)&#x2013;related management. Catheter care, daily activities, emergency, heparin flushing, insertion site abnormalities, and symptoms. Segments represent responses favoring model 1 (fine-tuned only; FT), model 2 (FT + retrieval-augmented generation [RAG]), and model 3 (FT + RAG + finite state machine [FSM]), with discordance (not determined) indicated separately.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92374_fig03.png"/></fig></sec><sec id="s3-2"><title>Query-Level Expert Evaluation Results</title><p>For the 43 query-level forced-choice preference evaluations across the three models, interrater agreement was moderate (Gwet AC1=0.581, 95% CI 0.417&#x2010;0.745; <italic>P</italic>&#x003C;.001). Cochran Q test showed significant differences in evaluator preference frequencies across the three models (Q<sub>2</sub>=48.2; <italic>P</italic>&#x003C;.001). Post hoc exact McNemar tests with Bonferroni correction indicated that model 3 was preferred significantly more often than model 1 (<italic>P</italic>&#x003C;.001) and model 2 (<italic>P</italic>&#x003C;.001), whereas model 1 and model 2 did not differ significantly (<italic>P</italic>=.999; Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). <xref ref-type="fig" rid="figure4">Figure 4</xref> presents a detailed expert preference analysis of all 43 queries. Among these, 23 (53%) achieved a unanimous consensus among the three experts, of which 20 (87%) favored model 3. This pattern indicates that the combined architecture incorporating FSM-based dialogue flow control and RAG-supported knowledge was the most frequently preferred option for the majority of clinically relevant queries.</p><p>However, for seven queries, at least one evaluator could not identify a preferred model during the initial evaluation and was subsequently requested to make an additional forced choice. This additional round resolved a determinate winner for six of these queries: Q5 (purchasing dressing supplies), Q6 (changing hospitals for PICC care), Q14 (managing numbness), Q15 (insertion site bleeding), Q20 (securing fixation device), and Q41 (sleeping with catheter). For the remaining query, Q30 (appropriate heparin flushing volume), the three models each received one vote and the outcome remained tied even after the additional forced choice.</p><p>Finally, interrater disagreement regarding model preference was observed for three queries: Q30, Q33 (use of the affected arm, physical activity, and daily task limitations), and Q34 (need for dressing replacement due to sweating after exercise). This variability suggests differences in clinical interpretations or context-dependent nuances within certain queries.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Expert selection of preferred models across queries. The circles represent each evaluator (N01, N02, and N03). Triangles (&#x25B3;) indicate queries for which an evaluator could not determine a preferred model initially and was subsequently requested to make an additional forced choice. IV: intravenous; PICC: peripherally inserted central catheters.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92374_fig04.png"/></fig></sec><sec id="s3-3"><title>Global Ratings of the Three Models</title><p>Across the eight evaluation criteria, model 3 received the highest mean ratings for most dimensions in this evaluation (<xref ref-type="fig" rid="figure5">Figure 5</xref>). Model 3&#x2019;s highest ratings were for Actionability (mean 4.67, SD 0.58) and Adaptability (mean 4.67, SD 0.58), while its comparatively lower ratings were for Efficiency (mean 3.33, SD 0.58) and Clarity (mean 3.67, SD 0.58; Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Compared with model 3, model 1 achieved a higher score for Efficiency (mean 4.00, SD 0.00 vs mean 3.33, SD 0.58). Model 1 showed intermediate ratings overall, whereas model 2 received the lowest ratings, particularly for Accuracy (mean 1.67, SD 0.58), Completeness (mean 1.67, SD 0.58), and Overall Quality (mean 1.67, SD 0.58).</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Comparison of global ratings across the 3 chatbot models on 8 evaluation dimensions. Radar chart visualization of expert global ratings (5-point scale) for model 1 (fine-tuned only; FT), model 2 (FT + retrieval-augmented generation [RAG]), and model 3 (FT + RAG + finite state machine [FSM]) across the 8 predefined evaluation dimensions. Each plotted value represents the mean of 3 independent evaluator ratings.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92374_fig05.png"/></fig></sec><sec id="s3-4"><title>Expert Debriefing on Key Principles for Improving LLM-Based PICC Consultation Chatbots</title><p>Through focus group interviews, experts considered three fundamental components during the quantitative evaluation: (1) in-depth assessment of the patient&#x2019;s problem and its underlying causes; (2) identification and efficient delivery of information that patients need; and (3) provision of accurate, evidence-based, and tailored responses. These elements were consistently emphasized throughout the discussion as essential criteria for evaluating chatbot performance and determining the appropriateness of consultation quality. The discordance among expert evaluations stems primarily from differing perspectives on the optimal balance between the comprehensiveness and efficiency of information delivery. Nevertheless, based on their rationale for high-quality PICC consultations, evaluators identified several notable strengths in model 3. The experts agreed that the FSM-guided dialogue flow was effective in delivering structured and reliable instructions in clinically careful situations.</p><disp-quote><p>The answers (from Model 3) were mostly based on recent evidence and accurately followed the guidelines.</p><attrib>N01</attrib></disp-quote><disp-quote><p>When patients use the chatbot for self-care or need to handle things on their own, I feel like model 2 doesn&#x2019;t really work that well. Model 1&#x2019;s answers were kind of vague too, so it was probably hard for patients to get clear information. And like she (N01) mentioned, both model 2 and model 1 sometimes didn&#x2019;t really answer the actual question &#x2014; they&#x2019;d kind of go off-topic or talk about something else instead.</p><attrib>N02</attrib></disp-quote><disp-quote><p>We&#x2019;ve used a rule-based chatbot before, but model 3 actually felt more like having a real conversation. There was definitely more of a sense of communicating back and forth.</p><attrib>N03</attrib></disp-quote><p>However, model 3 exhibited some limitations that tempered its overall effectiveness. The most prominent concern was the excessive response length with unnecessary ancillary information beyond the core message. In addition, model 3&#x2019;s assessment-oriented approach was appreciated for its depth, yet criticized for lacking actionable conclusions. Thus, to achieve an optimal balance between assessment depth and response efficiency, they recommended that the system secure more expert cases that include scenario-specific, detailed questioning, and decision-making.</p><disp-quote><p>It was like there were so many back-and-forths just to get to that answer. And in the end, it was just about giving the contact info.</p><attrib>N01</attrib></disp-quote><disp-quote><p>So with model 3, like I mentioned before, when the chatbot asks about a symptom and the patient says &#x201C;No,&#x201D; it just immediately moves on to something like, &#x201C;Since detailed consultation isn&#x2019;t possible here, please contact us directly.&#x201D; There&#x2019;s no real follow-up or explanation, so the patient might end up thinking &#x201C;Okay&#x2026; then why not just call right away?&#x201D;</p><attrib>N03</attrib></disp-quote></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study showed that a hybrid architecture integrating FT, RAG, and FSM was preferred by expert evaluators in the development of an LLM-based chatbot for PICC patient consultations. In blinded expert evaluations conducted by PICC-specialized nurses, model 3 (FT+RAG+FSM) achieved a preference rate of 79%, showing higher expert-rated acceptability over single-component architecture. In terms of global ratings, model 3 descriptively received higher Likert scores than the other two models on expert-rated actionability and adaptability.</p><p>The FSM-based dialogue management system in model 3 was implemented through a two-level hierarchical structure in which the major domains of PICC care, such as symptom assessment, daily management, and emergency conditions, were defined as high-level states. User queries were mapped to appropriate states via intent classification, after which low-level scenario flows generated responses that aligned with clinical protocols. This structure follows the design principles demonstrated in prior work, showing that decomposing complex task-oriented dialogue into subgoals and distributing decision-making across different levels improves response robustness and task completion [<xref ref-type="bibr" rid="ref32">32</xref>]. The established principles of medical consultation delineate a sequential framework comprising (1) problem identification, (2) comprehensive patient assessment, and (3) the provision of structured and tailored information [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. Considering that PICC consultations must adhere to these established principles, FSM-based response generation may support more clinically acceptable consultation by serving as a &#x201C;conversational guardrail&#x201D; that ensures structured, protocol-aligned interactions [<xref ref-type="bibr" rid="ref35">35</xref>]. Previous research has shown that in medical consultation scenarios, general LLMs used without procedural flow control based on clinical protocols may skip essential symptom assessment steps or generate clinically unverified recommendations [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. Similar patterns were observed in this study, as models without FSM-based control often deviated from the core clinical intent or produced unnecessary information, highlighting the need for structured flow management.</p><p>Contrary to our hypotheses, model 2 (fine-tuned model + RAG) received lower descriptive Likert ratings than model 1 (fine-tuned model alone) on overall quality and most evaluation dimensions. This may be because the contextual volume of the educational materials used in this study was insufficient. In general, the RAG was invented to address the hallucinations of LLMs; however, its performance can be affected by the quality of documents retrieved by LLMs [<xref ref-type="bibr" rid="ref38">38</xref>]. Notably, when documents contain insufficient contextual information to disambiguate vague or imprecise user queries, contradictory or distracting material is likely to be incorporated during the retrieval process, thereby diminishing the accuracy of LLM-generated responses [<xref ref-type="bibr" rid="ref39">39</xref>]. Previous research has demonstrated that the RAG system performance declined below that of the baseline model operating without a retrieved context when a single irrelevant document was incorporated into the retrieval set [<xref ref-type="bibr" rid="ref40">40</xref>]. Although numerous hospitals provide patients and caregivers with PICC education materials, the instructions are typically brief to ensure ease of understanding [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. Hence, it may have been relatively difficult for the RAG, which uses simplified documents, to provide quality and sufficient information based on the retrieved context (poor chunking quality). In addition, our retrieval was configured with top-k=3 without a similarity threshold or no-retrieval fallback. Hence, under sparse and brief source documents, this likely surfaced topically adjacent but non-gold chunks that were nonetheless injected into the prompt (poor retrieval relevance). The combination of these two mechanisms&#x2014;poor chunking quality and poor retrieval relevance&#x2014;may explain why model 2 fell below the model 1 baseline rather than defaulting to it. From this perspective, our findings suggest that integrating FSM frameworks into LLMs may mitigate existing technological limitations by establishing more deterministic conversational pathways tailored to individual patient inquiries, even when the underlying document repository is insufficient.</p><p>Despite its strengths, model 3 exhibited difficulties with several query types. These queries tend to involve context-dependent considerations tightly linked to real-world daily routines or individualized patient circumstances, domains in which flexible reasoning beyond the constraints of FSM rules is often required. Moreover, these cases frequently lack a single correct answer within the clinical guidelines, requiring experiential judgment from experts. Variability in expert interpretation, shaped by differing clinical priorities, such as emphasizing reassurance versus patient convenience, also contributed to disagreement in those cases. A related concern raised during expert evaluation was conversational efficiency. This is because the current FSM design elicits one piece of information per turn; clinically straightforward scenarios sometimes required more dialogue steps than necessary. A potential solution is multislot extraction, in which the LLM is prompted to identify multiple relevant entities (eg, symptom type and severity) within a single conversational turn, thereby reducing the number of required exchanges while preserving protocol adherence. This optimization, combined with adaptive turn-skipping logic that bypasses redundant assessment steps when sufficient information is already available, represents a promising direction for improving the practical usability of FSM-guided dialogue systems.</p><p>From a technical standpoint, the development of FSM-based states and transition rules requires substantial expertise and iterative refinement. Mapping diverse query types to predefined states and specifying appropriate transition conditions necessitate close collaboration between clinical experts and system developers, which may limit scalability and increase the system maintenance burden. To address these challenges, future systems may benefit from three key strategies: (1) proactive collection of expert-derived real-world cases, (2) transformation of these cases into formalized ontologies, and (3) the development of interpretable reasoning modules grounded in such ontologies.</p><p>Ontologies can structure hierarchical and semantic relationships between clinical concepts, partially automating intent classification and state mapping, while enhancing system extensibility [<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref43">43</xref>]. When integrated with standardized medical ontologies, the reliance on manually defined rules can be reduced while preserving clinical accuracy. Furthermore, ontology-based approaches offer reusable knowledge structures that support efficient scalability beyond PICC management to other clinical domains.</p></sec><sec id="s4-2"><title>Limitations</title><p>This study had several limitations. First, the evaluation was based on human expert assessments, and no benchmark dataset currently exists for validating PICC-related chatbot performance. The evaluation criteria were adapted from prior studies but were not derived from a formally validated instrument, limiting the ability to derive fully objective quantitative measures. Accordingly, future research should prioritize the development of dedicated evaluation metrics for LLM-mediated clinical consultations or, alternatively, adapt established frameworks, such as the Calgary&#x2013;Cambridge Guide, to assess adherence to the principles of effective patient&#x2013;physician communication [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>].</p><p>Second, although the assessment was conducted blinded, true blinding may have been severely compromised by the structurally distinct outputs of the FSM-guided model, which generated noticeably longer responses with multiturn back-and-forth interactions. While the evaluators received no prior briefing on the underlying technical mechanisms, the structural distinctness of the FSM-guided outputs may have inadvertently signaled the architecture in some cases. The expert preference for the FSM-guided model should therefore be interpreted with this caveat.</p><p>Third, model performance was constrained by the scope of the preconstructed educational materials used for training. Although these materials sufficiently cover core PICC management tasks, the system may be limited in responding to diverse real-world scenarios encountered in everyday life. Therefore, additional strategies are required to support flexible reasoning in unexpected and context-rich situations. In addition, the knowledge base and FSM transition rules were derived from educational materials and clinical workflows of a single tertiary cancer center. Institution-specific elements such as dressing intervals, role permissions for heparin flushing, and local escalation pathways may not generalize to settings with different staffing models or referral practices. The 43 evaluation queries, although derived from real patient inquiries, were collected at the same single cancer center, which may also limit their generalizability. Multicenter validation would be required before the proposed architecture can be considered clinically generalizable.</p><p>Fourth, the RAG chunking strategy used in this study relied on double-newline (paragraph-level) delimiters matched to our institutional materials, which were preformatted with consistent paragraph breaks. Given that PDF text extraction can introduce irregular or broken newline characters, this approach is sensitive to document formatting and may have reduced chunking quality. A more sophisticated, layout-aware parsing strategy might therefore have yielded different baseline results for model 2, and the chunking approach would need to be adapted before applying this architecture to external document sources.</p><p>Fifth, the evaluation relied solely on health care professionals and may not have fully reflected patient-perceived usefulness or usability. Establishing content validity by qualified clinical experts was a deliberate precondition of this study. It is crucial to ensure the safety of content before it is delivered to patients and to enable subsequent patient-experience data to be interpreted against a clinically validated baseline. Building on this expert-validated baseline, future work should extend the evaluation to include patient-facing usability assessment together with operational performance metrics such as system latency, token costs, and fallback frequency.</p></sec><sec id="s4-3"><title>Conclusions</title><p>In conclusion, an FSM-guided RAG architecture (model 3) was preferred by oncology nurses in 79% of PICC self-management scenarios over a fine-tuned&#x2013;only model (model 1) and a fine-tuned+RAG model (model 2). Model 3 received descriptively higher expert-rated acceptability scores on actionability, adaptability, and safety, while showing lower ratings on efficiency, possibly reflecting its multiturn structure. Model 2 received descriptively lower Likert ratings than model 1 on most dimensions, although the two single-component architectures did not differ significantly in 43-query model-preference rate after Bonferroni correction. These results indicate that retrieval augmentation alone, without an additional structural constraint, did not yield a robust improvement in expert-rated acceptability in this single-center evaluation. These findings support the incremental value of FSM-based dialogue control as a complement to FT and a RAG system. By constraining the conversation flow to state transitions based on clinical evidence, the chatbot&#x2019;s output becomes more closely aligned with established expert PICC consultation patterns. While this study did not assess patient-facing usability, our findings suggest the potential to provide more clinically plausible virtual consultations to patients.</p></sec></sec></body><back><ack><p>The authors used generative AI tools (ChatGPT and Claude, developed by OpenAI and Anthropic, respectively) to assist with code development for the chatbot system implementation. The authors take full responsibility for the accuracy and integrity of all content in this manuscript.</p></ack><notes><sec><title>Funding</title><p>This study was supported by the National Research Foundation of Korea (NRF) grant funded by the Korean government (MSIT; NRF-2021R1A2C2011083, RS-2023-00212647, and RS-2024-00440881) and a grant of the Korea Health Technology R&#x0026;D Project through the Korea Health Industry Development Institute (KHIDI), funded by the Ministry of Health and Welfare, Republic of Korea (RS-2024-00335937).</p></sec><sec><title>Data Availability</title><p>Data supporting the findings of this study are available upon request from the corresponding authors.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: ML, SB, KH, and JC.</p><p>Methodology: ML, SB, and KH.</p><p>Investigation: ML.</p><p>Data curation: ML and JY</p><p>Formal analysis: SB and JW.</p><p>Visualization: ML and SB.</p><p>Writing - original draft: ML and SB</p><p>Writing - review &#x0026; editing: all authors</p><p>Supervision: KH and JC</p><p>Funding acquisition: JC and ML</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">FSM</term><def><p>finite-state machine</p></def></def-item><def-item><term id="abb2">FT</term><def><p>fine-tuning</p></def></def-item><def-item><term id="abb3">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb4">PICC</term><def><p>peripherally inserted central catheter</p></def></def-item><def-item><term id="abb5">RAG</term><def><p>retrieval-augmented generation</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chopra</surname><given-names>V</given-names> </name><name name-style="western"><surname>Anand</surname><given-names>S</given-names> </name><name name-style="western"><surname>Krein</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Chenoweth</surname><given-names>C</given-names> </name><name name-style="western"><surname>Saint</surname><given-names>S</given-names> </name></person-group><article-title>Bloodstream infection, venous thrombosis, and peripherally inserted central catheters: reappraising the evidence</article-title><source>Am J Med</source><year>2012</year><month>08</month><volume>125</volume><issue>8</issue><fpage>733</fpage><lpage>741</lpage><pub-id pub-id-type="doi">10.1016/j.amjmed.2012.04.010</pub-id><pub-id pub-id-type="medline">22840660</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="web"><article-title>Peripherally inserted central catheter</article-title><source>National Cancer Institute</source><access-date>2024-12-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.cancer.gov/publications/dictionaries/cancer-terms/def/peripherally-inserted-central-catheter">https://www.cancer.gov/publications/dictionaries/cancer-terms/def/peripherally-inserted-central-catheter</ext-link></comment></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Pu</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Clinical characteristics of peripherally inserted central catheter-related complications in cancer patients undergoing chemotherapy: a prospective and observational study</article-title><source>BMC Cancer</source><year>2023</year><month>09</month><day>22</day><volume>23</volume><issue>1</issue><fpage>894</fpage><pub-id pub-id-type="doi">10.1186/s12885-023-11413-0</pub-id><pub-id pub-id-type="medline">37736715</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yap</surname><given-names>YS</given-names> </name><name name-style="western"><surname>Karapetis</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lerose</surname><given-names>S</given-names> </name><name name-style="western"><surname>Iyer</surname><given-names>S</given-names> </name><name name-style="western"><surname>Koczwara</surname><given-names>B</given-names> </name></person-group><article-title>Reducing the risk of peripherally inserted central catheter line complications in the oncology setting</article-title><source>Eur J Cancer Care (Engl)</source><year>2006</year><month>09</month><volume>15</volume><issue>4</issue><fpage>342</fpage><lpage>347</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2354.2006.00664.x</pub-id><pub-id pub-id-type="medline">16968315</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sharp</surname><given-names>R</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Pumpa</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Supportive care needs of adults living with a peripherally inserted central catheter (PICC) at home: a qualitative content analysis</article-title><source>BMC Nurs</source><year>2024</year><month>01</month><day>2</day><volume>23</volume><issue>1</issue><fpage>4</fpage><pub-id pub-id-type="doi">10.1186/s12912-023-01614-0</pub-id><pub-id pub-id-type="medline">38163877</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ma</surname><given-names>D</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>P</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>P</given-names> </name></person-group><article-title>Self-management of peripherally inserted central catheters after patient discharge via the WeChat smartphone application: a systematic review and meta-analysis</article-title><source>PLoS One</source><year>2018</year><volume>13</volume><issue>8</issue><fpage>e0202326</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0202326</pub-id><pub-id pub-id-type="medline">30153253</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zain</surname><given-names>NM</given-names> </name><name name-style="western"><surname>Yusuf</surname><given-names>A</given-names> </name></person-group><article-title>Home care-based education for cancer patients with peripherally inserted central catheters: a systematic review and meta-analysis</article-title><source>Support Care Cancer</source><year>2025</year><month>06</month><day>30</day><volume>33</volume><issue>7</issue><fpage>639</fpage><pub-id pub-id-type="doi">10.1007/s00520-025-09667-4</pub-id><pub-id pub-id-type="medline">40586980</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aggarwal</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tam</surname><given-names>CC</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><name name-style="western"><surname>Qiao</surname><given-names>S</given-names> </name></person-group><article-title>Artificial intelligence-based chatbots for promoting health behavioral changes: systematic review</article-title><source>J Med Internet Res</source><year>2023</year><month>02</month><day>24</day><volume>25</volume><fpage>e40789</fpage><pub-id pub-id-type="doi">10.2196/40789</pub-id><pub-id pub-id-type="medline">36826990</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sanders</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name><name name-style="western"><surname>Chow</surname><given-names>JCL</given-names> </name></person-group><article-title>Chatbot for health care and oncology applications using artificial intelligence and machine learning: systematic review</article-title><source>JMIR Cancer</source><year>2021</year><month>11</month><day>29</day><volume>7</volume><issue>4</issue><fpage>e27850</fpage><pub-id pub-id-type="doi">10.2196/27850</pub-id><pub-id pub-id-type="medline">34847056</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lam</surname><given-names>KT</given-names> </name><name name-style="western"><surname>Yip</surname><given-names>KM</given-names> </name><etal/></person-group><article-title>Comparison of an AI chatbot with a nurse hotline in reducing anxiety and depression levels in the general population: pilot randomized controlled trial</article-title><source>JMIR Hum Factors</source><year>2025</year><month>03</month><day>6</day><volume>12</volume><fpage>e65785</fpage><pub-id pub-id-type="doi">10.2196/65785</pub-id><pub-id pub-id-type="medline">40048637</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Therapeutic potential of social chatbots in alleviating loneliness and social anxiety: quasi-experimental mixed methods study</article-title><source>J Med Internet Res</source><year>2025</year><month>01</month><day>14</day><volume>27</volume><fpage>e65589</fpage><pub-id pub-id-type="doi">10.2196/65589</pub-id><pub-id pub-id-type="medline">39808786</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Safranek</surname><given-names>C</given-names> </name><name name-style="western"><surname>Socrates</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Patient-representing population&#x2019;s perceptions of GPT-generated versus standard emergency department discharge instructions: randomized blind survey assessment</article-title><source>J Med Internet Res</source><year>2024</year><month>08</month><day>2</day><volume>26</volume><fpage>e60336</fpage><pub-id pub-id-type="doi">10.2196/60336</pub-id><pub-id pub-id-type="medline">39094112</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Fan</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Ability of ChatGPT to replace doctors in patient education: cross-sectional comparative analysis of inflammatory bowel disease</article-title><source>J Med Internet Res</source><year>2025</year><month>03</month><day>31</day><volume>27</volume><fpage>e62857</fpage><pub-id pub-id-type="doi">10.2196/62857</pub-id><pub-id pub-id-type="medline">40163853</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sallam</surname><given-names>M</given-names> </name></person-group><article-title>ChatGPT utility in healthcare education, research, and practice: systematic review on the promising perspectives and valid concerns</article-title><source>Health Care (Don Mills)</source><year>2023</year><month>03</month><day>19</day><volume>11</volume><issue>6</issue><fpage>887</fpage><pub-id pub-id-type="doi">10.3390/healthcare11060887</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>McTear</surname><given-names>M</given-names> </name><name name-style="western"><surname>Varghese Marokkie</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bi</surname><given-names>Y</given-names> </name></person-group><article-title>A comparative study of chatbot response generation: traditional approaches versus large language models</article-title><year>2023</year><conf-name>Knowledge Science, Engineering and Management: 16th International Conference, KSEM 2023</conf-name><conf-date>Aug 16-18, 2023</conf-date><conf-loc>Guangzhou, China</conf-loc><fpage>70</fpage><lpage>79</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-40286-9_7</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ayala</surname><given-names>O</given-names> </name><name name-style="western"><surname>Bechard</surname><given-names>P</given-names> </name></person-group><article-title>Reducing hallucination in structured outputs via retrieval-augmented generation</article-title><conf-name>Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics</conf-name><conf-date>Apr 16-21, 2024</conf-date><conf-loc>Mexico City, Mexico</conf-loc><fpage>228</fpage><lpage>238</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.naacl-industry.19</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>M</given-names> </name><name name-style="western"><surname>He</surname><given-names>B</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Mitigating large language model hallucination with faithful finetuning</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 17, 2024</comment><pub-id pub-id-type="doi">10.48550/arxiv.2406.11267</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name></person-group><article-title>Hallucination mitigation for retrieval-augmented large language models: a review</article-title><source>Mathematics</source><year>2025</year><volume>13</volume><issue>5</issue><fpage>856</fpage><pub-id pub-id-type="doi">10.3390/math13050856</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="web"><article-title>How to care for your PICC line at home</article-title><source>Monument Health</source><year>2025</year><month>03</month><day>7</day><access-date>2026-07-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://youreducation.elsevier.com/display/english/document/ad41ffd8-58e5-4037-a4d7-204959513652/95253918">https://youreducation.elsevier.com/display/english/document/ad41ffd8-58e5-4037-a4d7-204959513652/95253918</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="web"><article-title>About your peripherally inserted central catheter (PICC)</article-title><source>Memorial Sloan Kettering Cancer Center</source><access-date>2025-03-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.mskcc.org/cancer-care/patient-education/about-your-peripherally-inserted-central-catheter-picc">https://www.mskcc.org/cancer-care/patient-education/about-your-peripherally-inserted-central-catheter-picc</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="web"><article-title>Peripherally inserted central catheter (PICC) line</article-title><source>Mayo Clinic</source><year>2025</year><month>03</month><day>7</day><access-date>2026-07-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.mayoclinic.org/tests-procedures/picc-line/about/pac-20468748">https://www.mayoclinic.org/tests-procedures/picc-line/about/pac-20468748</ext-link></comment></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="web"><article-title>Peripherally inserted central catheter (PICC): care instructions</article-title><source>Kaiser Permanente</source><access-date>2025-03-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://healthy.kaiserpermanente.org/health-wellness/health-encyclopedia/he.peripherally-inserted-central-catheter-picc-care-instructions.ug6122">https://healthy.kaiserpermanente.org/health-wellness/health-encyclopedia/he.peripherally-inserted-central-catheter-picc-care-instructions.ug6122</ext-link></comment></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qi</surname><given-names>W</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>G</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>T</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name></person-group><article-title>Effects of self-management education integrated nursing on cancer patients with PICC placement: a systematic review and meta-analysis</article-title><source>J Res Nurs</source><year>2024</year><month>11</month><volume>29</volume><issue>7</issue><fpage>515</fpage><lpage>532</lpage><pub-id pub-id-type="doi">10.1177/17449871241268513</pub-id><pub-id pub-id-type="medline">39544446</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>P</given-names> </name><name name-style="western"><surname>Perez</surname><given-names>E</given-names> </name><name name-style="western"><surname>Piktus</surname><given-names>A</given-names> </name><name name-style="western"><surname>Petroni</surname><given-names>F</given-names> </name><name name-style="western"><surname>Karpukhin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Goyal</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation for knowledge-intensive NLP tasks</article-title><year>2020</year><conf-name>34th Conference on Neural Information Processing Systems (NeurIPS 2020)</conf-name><conf-date>Dec 6-12, 2020</conf-date><pub-id pub-id-type="doi">10.48550/arXiv.2005.11401</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Pang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bi</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Development and evaluation of a retrieval-augmented large language model framework for ophthalmology</article-title><source>JAMA Ophthalmol</source><year>2024</year><month>09</month><day>1</day><volume>142</volume><issue>9</issue><fpage>798</fpage><lpage>805</lpage><pub-id pub-id-type="doi">10.1001/jamaophthalmol.2024.2513</pub-id><pub-id pub-id-type="medline">39023885</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Raux</surname><given-names>A</given-names> </name><name name-style="western"><surname>Eskenazi</surname><given-names>M</given-names> </name></person-group><article-title>A finite-state turn-taking model for spoken dialog systems</article-title><conf-name>Human Language Technologies: Conference of the North American Chapter of the Association of Computational Linguistics, Proceedings</conf-name><conf-date>May 31 to Jun 5, 2009</conf-date><pub-id pub-id-type="doi">10.3115/1620754.1620846</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lynn</surname><given-names>MR</given-names> </name></person-group><article-title>Determination and quantification of content validity</article-title><source>Nurs Res</source><year>1986</year><volume>35</volume><issue>6</issue><fpage>382</fpage><lpage>385</lpage><pub-id pub-id-type="doi">10.1097/00006199-198611000-00017</pub-id><pub-id pub-id-type="medline">3640358</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Polit</surname><given-names>DF</given-names> </name><name name-style="western"><surname>Beck</surname><given-names>CT</given-names> </name></person-group><article-title>The content validity index: are you sure you know what&#x2019;s being reported? Critique and recommendations</article-title><source>Res Nurs Health</source><year>2006</year><month>10</month><volume>29</volume><issue>5</issue><fpage>489</fpage><lpage>497</lpage><pub-id pub-id-type="doi">10.1002/nur.20147</pub-id><pub-id pub-id-type="medline">16977646</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tam</surname><given-names>TYC</given-names> </name><name name-style="western"><surname>Sivarajkumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kapoor</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A framework for human evaluation of large language models in healthcare derived from literature review</article-title><source>NPJ Digit Med</source><year>2024</year><month>09</month><day>28</day><volume>7</volume><issue>1</issue><fpage>258</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01258-7</pub-id><pub-id pub-id-type="medline">39333376</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jo</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>MJ</given-names> </name><etal/></person-group><article-title>User experiences of a chatbot for supporting the self-management of peripherally inserted central catheter for chemotherapy: mixed methods study</article-title><source>JMIR Cancer</source><year>2026</year><month>02</month><day>11</day><volume>12</volume><fpage>e81026</fpage><pub-id pub-id-type="doi">10.2196/81026</pub-id><pub-id pub-id-type="medline">41671551</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wongpakaran</surname><given-names>N</given-names> </name><name name-style="western"><surname>Wongpakaran</surname><given-names>T</given-names> </name><name name-style="western"><surname>Wedding</surname><given-names>D</given-names> </name><name name-style="western"><surname>Gwet</surname><given-names>KL</given-names> </name></person-group><article-title>A comparison of Cohen&#x2019;s Kappa and Gwet&#x2019;s AC1 when calculating inter-rater reliability coefficients: a study conducted with personality disorder samples</article-title><source>BMC Med Res Methodol</source><year>2013</year><month>04</month><day>29</day><volume>13</volume><issue>61</issue><fpage>61</fpage><pub-id pub-id-type="doi">10.1186/1471-2288-13-61</pub-id><pub-id pub-id-type="medline">23627889</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Mo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Maharaj</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hishamunda</surname><given-names>B</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name></person-group><article-title>HierTOD: a task-oriented dialogue system driven by hierarchical goals</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 11, 2024</comment><pub-id pub-id-type="doi">10.3390/math11143048</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Iversen</surname><given-names>ED</given-names> </name><name name-style="western"><surname>Wolderslund</surname><given-names>MO</given-names> </name><name name-style="western"><surname>Kofoed</surname><given-names>PE</given-names> </name><etal/></person-group><article-title>Codebook for rating clinical communication skills based on the Calgary-Cambridge Guide</article-title><source>BMC Med Educ</source><year>2020</year><month>05</month><day>6</day><volume>20</volume><issue>1</issue><fpage>140</fpage><pub-id pub-id-type="doi">10.1186/s12909-020-02050-3</pub-id><pub-id pub-id-type="medline">32375756</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burt</surname><given-names>J</given-names> </name><name name-style="western"><surname>Abel</surname><given-names>G</given-names> </name><name name-style="western"><surname>Elmore</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Assessing communication quality of consultations in primary care: initial reliability of the Global Consultation Rating Scale, based on the Calgary-Cambridge Guide to the medical interview</article-title><source>BMJ Open</source><year>2014</year><month>03</month><day>6</day><volume>4</volume><issue>3</issue><fpage>e004339</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2013-004339</pub-id><pub-id pub-id-type="medline">24604483</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hakim</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Painter</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Ramcharran</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kara</surname><given-names>V</given-names> </name><name name-style="western"><surname>Powell</surname><given-names>G</given-names> </name><name name-style="western"><surname>Sobczak</surname><given-names>P</given-names> </name><etal/></person-group><article-title>The need for guardrails with large language models in medical safety-critical settings: an artificial intelligence application in the pharmacovigilance ecosystem</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 4, 2024</comment><pub-id pub-id-type="doi">10.1038/s41598-025-09138-0</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ren</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>L</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Tao</surname><given-names>D</given-names> </name></person-group><article-title>Healthcare agent: eliciting the power of large language models for medical consultation</article-title><source>npj Artif Intell</source><year>2025</year><volume>1</volume><issue>1</issue><fpage>24</fpage><pub-id pub-id-type="doi">10.1038/s44387-025-00021-x</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Shi</surname><given-names>X</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Du</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Medical dialogue system: a survey of categories, methods, evaluation and challenges</article-title><conf-name>Findings of the Association for Computational Linguistics ACL 2024</conf-name><conf-date>Aug 11-16, 2024</conf-date><conf-loc>Bangkok, Thailand</conf-loc><fpage>2840</fpage><lpage>2861</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.findings-acl.167</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yan</surname><given-names>SQ</given-names> </name><name name-style="western"><surname>Gu</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ling</surname><given-names>ZH</given-names> </name></person-group><article-title>Corrective retrieval augmented generation</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 29, 2024</comment><pub-id pub-id-type="doi">10.2139/ssrn.5267341</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Barnett</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kurniawan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Thudumu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Brannelly</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Abdelrazek</surname><given-names>M</given-names> </name></person-group><article-title>Seven failure points when engineering a retrieval augmented generation system</article-title><conf-name>Proceedings of the IEEE/ACM 3rd International Conference on AI Engineering-Software Engineering for AI</conf-name><conf-date>Apr 14-15, 2024</conf-date><conf-loc>Lisbon Portugal</conf-loc><fpage>194</fpage><lpage>199</lpage><pub-id pub-id-type="doi">10.1145/3644815.3644945</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Cuconasu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Trappolini</surname><given-names>G</given-names> </name><name name-style="western"><surname>Siciliano</surname><given-names>F</given-names> </name><etal/></person-group><article-title>The power of noise: redefining retrieval for rag systems</article-title><conf-name>SIGIR 2024</conf-name><conf-date>Jul 14-18, 2024</conf-date><conf-loc>Washington DC USA</conf-loc><fpage>719</fpage><lpage>729</lpage><pub-id pub-id-type="doi">10.1145/3626772.3657834</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Al-Zubaide</surname><given-names>H</given-names> </name><name name-style="western"><surname>Issa</surname><given-names>AA</given-names> </name></person-group><article-title>OntBot: ontology based chatbot</article-title><year>2011</year><conf-name>International Symposium on Innovations in Information and Communications Technology</conf-name><conf-date>Nov 29 to Dec 1, 2011</conf-date><conf-loc>Amman, Jordan</conf-loc><fpage>7</fpage><lpage>12</lpage><pub-id pub-id-type="doi">10.1109/ISIICT.2011.6149594</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>R</given-names> </name><name name-style="western"><surname>Croxford</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Leveraging medical knowledge graphs into large language models for diagnosis prediction: design and application study</article-title><source>JMIR AI</source><year>2025</year><month>02</month><day>24</day><volume>4</volume><fpage>e58670</fpage><pub-id pub-id-type="doi">10.2196/58670</pub-id><pub-id pub-id-type="medline">39993309</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tian</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lyu</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Electronic health record-oriented knowledge graph system for collaborative clinical decision support using multicenter fragmented medical data: design and application study</article-title><source>J Med Internet Res</source><year>2024</year><month>07</month><day>5</day><volume>26</volume><fpage>e54263</fpage><pub-id pub-id-type="doi">10.2196/54263</pub-id><pub-id pub-id-type="medline">38968598</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>.Supplemental materials.</p><media xlink:href="jmir_v28i1e92374_app1.docx" xlink:title="DOCX File, 41 KB"/></supplementary-material></app-group></back></article>