<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.0" xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMIR</journal-id>
      <journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id>
      <journal-title>Journal of Medical Internet Research</journal-title>
      <issn pub-type="epub">1438-8871</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v28i1e98131</article-id>
      <article-id pub-id-type="pmid"/>
      <article-id pub-id-type="doi">10.2196/98131</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Safety-Oriented Benchmarking of Large Language Models in Risk-Based Management of Abnormal Cervical Screening Results: Scenario-Based Benchmark Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Balcarras</surname>
            <given-names>Matthew</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Taiwo</surname>
            <given-names>Peter</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Yu</surname>
            <given-names>Yunguo</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Eroğlu</surname>
            <given-names>Ömer Osman</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Department of Obstetrics and Gynecology</institution>
            <institution>Ankara Etlik City Hospital</institution>
            <addr-line>Varlık Mahallesi, Halil Sezai Erkut Caddesi No:5</addr-line>
            <addr-line>Ankara, 06170</addr-line>
            <country>Türkiye</country>
            <phone>90 5428161338</phone>
            <email>omerosmaneroglu@gmail.com</email>
          </address>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-9861-6767</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Eroğlu</surname>
            <given-names>Cansın</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0004-5049-1415</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Department of Obstetrics and Gynecology</institution>
        <institution>Ankara Etlik City Hospital</institution>
        <addr-line>Ankara</addr-line>
        <country>Türkiye</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Ömer Osman Eroğlu <email>omerosmaneroglu@gmail.com</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>22</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>28</volume>
      <elocation-id>e98131</elocation-id>
      <history>
        <date date-type="received">
          <day>13</day>
          <month>4</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>10</day>
          <month>8</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>9</day>
          <month>9</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>9</day>
          <month>9</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Ömer Osman Eroğlu, Cansın Eroğlu. Originally published in the Journal of Medical Internet Research (https://www.jmir.org), 22.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on https://www.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://www.jmir.org/2026/1/e98131" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Large language models (LLMs) are increasingly being considered for clinical decision support, yet their safety in risk-based cervical screening management remains insufficiently characterized.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study benchmarked the guideline concordance and safety-related performance of 3 LLMs in the initial American Society for Colposcopy and Cervical Pathology (ASCCP) risk-based management of abnormal cervical screening results, using a purposively constructed synthetic scenario set that oversamples complex and history-dependent decision nodes.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>We developed 60 synthetic clinical scenarios reflecting initial abnormal screening management in immunocompetent, nonpregnant women aged 25 to 65 years using a predefined scenario coverage matrix. GPT-5.3, Gemini 3 Flash, and DeepSeek V3.2 were tested under 2 prompt conditions: Baseline Clinical Prompt and Guideline-Directed Prompt Package. Each scenario was run in 3 independent repetitions per model and prompt arm (1080 total observations). Responses were evaluated by 2 obstetrics and gynecology specialists, blinded to model and prompt-arm identity but not independent of gold-standard construction, using a prespecified rubric. The primary end point was the unsafe major error-free rate. Proportions are reported with CIs adjusted for within-scenario clustering, and generalized estimating equations were used for inferential comparisons.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>Under the guideline-directed prompt package, the unsafe major error-free rate was 100% (95% CI 94%-100%) for GPT-5.3, 98.9% (95% CI 94%-99.8%) for Gemini 3 Flash, and 75% (95% CI 63.2%-84%) for DeepSeek V3.2. In the main-effects model, the guideline-directed prompt package was associated with higher odds of both unsafe major error-free performance (odds ratio [OR] 3.76, 95% CI 2.45-5.77; <italic>P</italic>&lt;.001) and exact concordance (OR 8.27, 95% CI 4.14-16.52; <italic>P</italic>&lt;.001). Error rates increased substantially with scenario complexity, rising from 3.1% in low-complexity to 29% in high-complexity scenarios. The most frequent error subtypes were undermanagement, genotype misinterpretation, and history neglect. Interrater agreement was almost perfect (weighted κ=0.839, 95% CI 0.811-0.867).</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>Safety-oriented benchmark performance in initial ASCCP risk-based management varied markedly by model, prompt condition, and scenario complexity. The guideline-directed prompt package was associated with improvement in both safety and guideline concordance, but even the best-performing model remained vulnerable in complex, history-dependent scenarios. LLMs may have value as clinician-supervised decision support tools, but these benchmark findings should not be interpreted as supporting autonomous clinical use in cervical screening management.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>American Society for Colposcopy and Cervical Pathology</kwd>
        <kwd>ASCCP guidelines</kwd>
        <kwd>ASCCP</kwd>
        <kwd>benchmark study</kwd>
        <kwd>cervical cancer screening</kwd>
        <kwd>clinical decision support</kwd>
        <kwd>guideline concordance</kwd>
        <kwd>large language model</kwd>
        <kwd>LLM</kwd>
        <kwd>patient safety</kwd>
        <kwd>prompt engineering</kwd>
        <kwd>risk-based management</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>Cervical cancer screening is one of the most successful cancer prevention strategies in women’s health; however, its clinical effectiveness depends not only on the performance of screening tests but also on the accurate and timely management of abnormal results [<xref ref-type="bibr" rid="ref1">1</xref>]. In this context, the risk-based management consensus guidelines published by the American Society for Colposcopy and Cervical Pathology (ASCCP) in 2019 introduced a major paradigm shift in the management of abnormal cervical screening results [<xref ref-type="bibr" rid="ref1">1</xref>]. Unlike earlier algorithm-based approaches, these guidelines established a decision framework based on the estimated risk of cervical intraepithelial neoplasia grade 3 or worse (CIN 3+), integrating the current test result with prior screening history, HPV-based testing, genotype information, colposcopic and biopsy findings, and treatment history [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Six distinct clinical action thresholds were defined, and management decisions are determined according to these thresholds [<xref ref-type="bibr" rid="ref1">1</xref>]. The guidelines were subsequently revised through formal updates [<xref ref-type="bibr" rid="ref4">4</xref>] and evolved under the Enduring Consensus Guidelines framework [<xref ref-type="bibr" rid="ref5">5</xref>], a standing process in which recommendations are revised continuously as new evidence emerges rather than through periodic wholesale revision. The resulting decision environment is therefore dynamic: management thresholds and triage options may change between formal guideline editions. The integration of emerging triage tools, such as dual stain testing, into the same risk-threshold architecture further illustrates that cervical screening management is becoming an increasingly dynamic decision domain [<xref ref-type="bibr" rid="ref6">6</xref>].</p>
      <p>Although the ASCCP risk-based framework offers more precise and individualized clinical management, it has also substantially increased the complexity of the decision-making process. The same current screening result may require entirely different initial management depending on prior human papillomavirus (HPV)–based test results, colposcopy history, or histopathologic background [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. In particular, history-dependent decision nodes, genotype-sensitive distinctions involving HPV 16/18, borderline colposcopy thresholds, and expedited treatment decisions transform this domain from a simple algorithmic application into one requiring contextual clinical reasoning [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. This complexity poses a significant challenge not only for AI systems but also for clinicians. Indeed, the ASCCP Cervical Cancer Screening Task Force has reported difficulties in achieving guideline-concordant practice during the adoption of new screening strategies and noted that the transition period in clinical practice has been longer than anticipated [<xref ref-type="bibr" rid="ref7">7</xref>].</p>
      <p>Large language models (LLMs) have rapidly expanded across diverse health care applications in recent years, including information access, patient education, clinical documentation, and decision support [<xref ref-type="bibr" rid="ref8">8</xref>]. However, apparently high performance on medical tasks is not equivalent to clinically safe behavior. Contemporary benchmark literature emphasizes that multiple-choice or short-answer tests do not adequately reflect real-world clinical performance and that open-ended, scenario-based, and safety-oriented evaluation frameworks are more informative [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Wang et al [<xref ref-type="bibr" rid="ref10">10</xref>] evaluated 6 LLMs within a dual framework of clinical safety and effectiveness and demonstrated that performance declined by an average of 13.3% in high-risk clinical scenarios. Multiple-choice and closed-format evaluations, by contrast, may overestimate readiness for open-ended clinical decision-making, since they constrain the response space and do not require the model to construct a management plan [<xref ref-type="bibr" rid="ref11">11</xref>]. Gaber et al [<xref ref-type="bibr" rid="ref9">9</xref>] benchmarked LLM workflows, including a retrieval-augmented configuration, on 2000 intensive care cases and found that the models could suggest likely diagnoses, appropriate specialists, and urgency of care, while continuing to struggle with nuanced clinical data. Ong et al [<xref ref-type="bibr" rid="ref12">12</xref>] tested an LLM as a medication safety clinical decision support system across 16 clinical specialties and demonstrated that the model improved the accuracy of medication chart review when used alongside pharmacists, compared with either the model or the pharmacist working alone. Together, these findings suggest that LLMs should assume an assistive rather than autonomous role in clinical decision support. Furthermore, recent multirun studies have shown that accuracy and consistency can behave independently when the same inputs are repeatedly presented to the same model and that high accuracy does not necessarily correspond to high stability [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>].</p>
      <p>Only a limited number of studies have evaluated LLM performance in the field of cervical cancer screening and management. Yurtcu et al [<xref ref-type="bibr" rid="ref15">15</xref>] assessed ChatGPT’s responses to frequently asked questions about cervical cancer and reported acceptable performance at a general knowledge level, while demonstrating that accuracy declined for guideline-based questions. Kuerbanjiang et al [<xref ref-type="bibr" rid="ref16">16</xref>] tested 9 different LLMs using 100 standardized questions related to cervical cancer management and observed that prompt engineering could improve performance; however, they combined 5 different guidelines and did not use history-dependent scenario design. Angyal et al [<xref ref-type="bibr" rid="ref17">17</xref>] developed a customized GPT model as a patient education tool in cervical cancer screening and reported favorable usability outcomes, but did not assess clinician-level decision safety. Pavone et al [<xref ref-type="bibr" rid="ref18">18</xref>] tested ChatGPT 4.0, DeepSeek R1, and Gemini 2.0 against European Society of Gynaecological Oncology (ESGO), European Society for Radiotherapy and Oncology (ESTRO), and European Society of Pathology (ESP) guidelines using 50 questions and reported that all models were inadequate in guideline concordance; however, this study did not include prompt comparison, did not classify error subtypes, and did not employ a safety-centered primary end point.</p>
      <p>These prior studies share several common limitations: most used accuracy-focused simple question-and-answer formats, did not apply systematic scenario design based on a single guideline, did not test history-dependent scenarios in which prior screening history alters management, did not classify error profiles from a clinical safety perspective, and did not evaluate the effect of prompt strategy on performance in a controlled manner. To our knowledge, no benchmark study has specifically focused on the 2019 ASCCP risk-based initial management approach, systematically sampled history-dependent decision nodes, employed a clinical safety-centered primary end point, and evaluated the effect of the guideline-directed prompt package in a controlled design.</p>
      <p>In the present study, we comparatively evaluated the guideline concordance and safety-oriented benchmark performance of 3 publicly accessible LLMs (GPT-5.3, Gemini 3 Flash, and DeepSeek V3.2) in the initial management of abnormal cervical screening results arising in an immunocompetent, nonpregnant, routine screening population, using 60 predefined synthetic clinical scenarios based on the 2019 ASCCP risk-based management guidelines. The primary end point was the unsafe major error-free rate, and the main secondary end point was the rate of exact concordance with the prespecified gold-standard management decision. The study additionally aimed to examine the effect of a guideline-directed prompt package strategy on model performance, the distribution of errors according to scenario complexity, the clinical patterns of error subtypes, and intramodel consistency.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Ethical Considerations</title>
        <p>The study was conducted within the Department of Obstetrics and Gynecology, Ankara Etlik City Hospital. The Clinical Research Ethics Committee of Ankara Etlik City Hospital granted an ethics exemption based on the exemption petition filed by the principal investigator (reference number 308763928; March 18, 2026), on the grounds that the study did not involve real patient data, biological specimens, or human participants. The overall study flow is summarized in <xref rid="figure1" ref-type="fig">Figure 1</xref>.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Study flow from scenario construction and ethics exemption through model querying, independent evaluation, 3-tier classification, and outcome derivation. ASCCP: American Society for Colposcopy and Cervical Pathology.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e98131_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Study Design</title>
        <p>This study was designed as a scenario-based, comparative, observational benchmark study to evaluate the guideline concordance and safety-oriented benchmark performance of LLMs in the initial management of abnormal cervical screening results according to the 2019 ASCCP risk-based approach. No real patient data were used; all inputs consisted of synthetic but clinically realistic scenarios developed on the basis of a predefined scenario coverage matrix. The study design is consistent with the contemporary benchmark literature, which emphasizes that the evaluation of open-ended clinical decision tasks should assess not only accuracy but also safety and consistency as separate dimensions [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref14">14</xref>].</p>
        <p>The study focused exclusively on initial abnormal screening management. Postcolposcopy surveillance, posttreatment follow-up, pregnancy, immunosuppression, HIV infection, transplant recipients, and other special high-risk subpopulations were excluded from the scope. The target clinical framework was the initial management of abnormal cervical screening results arising in an immunocompetent, nonpregnant, routine cervical cancer screening population aged 25 to 65 years [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref6">6</xref>].</p>
      </sec>
      <sec>
        <title>Gold Standard and Guideline Framework</title>
        <p>Scenario-level answer keys were anchored to the 2019 ASCCP Risk-Based Management Consensus Guidelines and the supporting studies that established the risk-estimation framework underlying these guidelines [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Official updates through 2023 were reviewed for currency during post hoc source verification. One scenario concerns the repeat interval for unsatisfactory cytology, for which the prespecified answer key reproduces the 2019 wording; the 2023 update subsequently removed the minimum 2-month interval while retaining the 4-month upper limit. Because the 2023 revision removed the minimum 2-month interval while retaining the 4-month upper limit, it permits earlier rather than later repeat testing. All unsafe major errors recorded for this scenario were classified as undermanagement, which reflects delayed or omitted escalation rather than an interval that was too short; applying the updated wording would therefore not have altered any classification in this dataset. The enduring consensus process and dual-stain recommendations were considered as contextual framework materials and did not alter any answer key. The prespecified decision for each of the 60 scenarios, together with its ASCCP source reference, is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <p>For each scenario, the accepted correct initial management decision was prespecified according to the ASCCP clinical action thresholds based on CIN 3+ risk. Within this framework, the following management options were used as reference: treatment, situations in which either treatment or colposcopy or biopsy was acceptable, colposcopy or biopsy, 1-year surveillance, 3-year surveillance, and return to routine screening at 5 years [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>].</p>
      </sec>
      <sec>
        <title>Prespecified Evaluation Rubric</title>
        <p>To minimize evaluator subjectivity, a scenario-level scoring workbook was prepared before data collection. For each scenario, it specified the scenario code and complexity level, the key clinical inputs, the relevant ASCCP decision node, and the accepted correct initial management decision. Minor and major deviations and their subtypes were assigned by consensus during evaluation using the operational categories available in the scoring workbook; a separate written adjudication manual was not created, and this is reported as a limitation. This structure was designed to distinguish clinically safe but imperfect responses from major errors carrying clinically meaningful harm potential [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref14">14</xref>]. The accepted correct decision recorded in the workbook is the preferred management option used for exact-concordance scoring. Where the guideline states that 2 options are equally acceptable, both were recorded and both were scored as exact correct. Where the guideline names a preferred option together with an acceptable alternative, only the preferred option was recorded; a response selecting the acceptable alternative was classified as safe but imperfect rather than as an error.</p>
      </sec>
      <sec>
        <title>Development of Clinical Scenarios</title>
        <p>A total of 60 synthetic clinical scenarios were developed on the basis of a predefined scenario coverage matrix. Scenarios were not selected at random; rather, they were systematically structured to cover the key decision nodes of ASCCP risk-based initial management. Scenarios were divided into 2 main groups: core scenarios (n=48) and challenge scenarios (n=12).</p>
        <p>Core scenarios were designed to systematically sample the major decision nodes of ASCCP risk-based initial management. Scenario development was based on combinations of the following variables: age group, HPV result (negative, positive non–16/18, HPV 16 positive, and HPV 18 positive), cytology result (negative for intraepithelial lesion or malignancy [NILM], atypical squamous cells of undetermined significance [ASC-US], low-grade squamous intraepithelial lesion [LSIL], atypical squamous cells, cannot exclude high-grade squamous intraepithelial lesion [ASC-H], high-grade squamous intraepithelial lesion [HSIL], and atypical glandular cells [AGC]), prior screening history (unknown, negative HPV-based test, and negative cytology alone), and prior biopsy or treatment history. This approach was designed to reflect the ASCCP decision logic, in which the current result is interpreted not in isolation but together with prior history and risk context [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>].</p>
        <p>Challenge scenarios were specifically designed to test situations in which models were expected to encounter difficulty, including scenarios in which the same current result required different management depending on prior history, genotype-specific exception rules involving HPV 16/18, combinations approaching the expedited treatment threshold, age-dependent decision distinctions, and cases in which prior screening history consisted only of cytology-based testing.</p>
        <p>The complete list of scenarios is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p>
      </sec>
      <sec>
        <title>Scenario Complexity Classification</title>
        <p>Each scenario was assigned a priori to 1 of 3 complexity levels. Low complexity (n=16) represented standard guideline application scenarios in which correct management could largely be determined from the current test result alone. Medium complexity (n=21) represented scenarios requiring integration of at least one prior clinical variable in addition to the current result. High complexity (n=23) represented scenarios involving guideline exceptions, history-dependent management, or multistep decision processes. This classification was used to analyze whether model performance varied not only at an overall level but also according to the difficulty of the clinical decision structure [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref14">14</xref>].</p>
      </sec>
      <sec>
        <title>Models Evaluated</title>
        <p>Three LLMs were evaluated: GPT-5.3 (OpenAI; released March 3, 2026), Gemini 3 Flash (Google DeepMind; released December 17, 2025), and DeepSeek V3.2 (DeepSeek AI; released December 1, 2025). Final model selection was based on contemporaneous availability during the testing window, free public accessibility through web interfaces, and representation of 3 different model providers. All models were tested using the most current versions accessible during the study period.</p>
        <p>Models were accessed through the publicly available web interface of each provider. No API was used. Each query was initiated in an independent, non–account-linked session, so no conversational context was carried over from prior sessions. No explicit web search was invoked, and no response contained source citations indicating retrieval. All models were tested in their free publicly available versions; as no user-configurable reasoning mode was available in the tested interfaces, default settings were used. Scenario development, gold-standard determination, data collection, response evaluation, statistical analysis, and clinical interpretation were performed entirely by the investigators.</p>
      </sec>
      <sec>
        <title>Prompt Design</title>
        <p>Each scenario was run in a new and independent session for each model. No prior conversational context, memory, or sequential feedback was carried between sessions. Only the first response was analyzed. The study included 2 main prompt arms.</p>
        <p>Arm 1 (baseline clinical prompt): the clinical scenario was presented directly to the model without any guideline name, directive phrasing, or structured output instruction.</p>
        <p>Arm 2 (guideline-directed prompt package): The same clinical scenario was presented together with 3 components that were varied jointly rather than separately: the 2019 ASCCP framework was named explicitly, the model was directed to apply the risk-based approach, and a structured output format was requested. Because these 3 components were introduced together, the design cannot separate the contribution of output structure from that of naming the guideline; the arm is therefore described throughout as a multicomponent package.</p>
        <p>In addition, an exploratory third arm (arm 3: checklist-based self-verification) was applied to a subset of 12 challenge scenarios. In this arm, the model first generated a response using the arm 2 prompt, after which a structured checklist was presented requesting the model to review and, if necessary, correct its response. This arm was not included in the primary model comparison and was reported solely as exploratory. For arm 3, the arm 2 prompt was rerun in separate independent sessions; these 108 responses were additional to, rather than a subset of, the 1080 primary-analysis responses. The verbatim prompt templates for all 3 arms are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p>
        <p>At the end of each prompt, the model was instructed to provide its response in plain text and to conclude with the format “Final Recommendation: [...].” Because the effect of prompt strategy on performance has been reported in prior cervical cancer and general medical LLM studies, the controlled comparison of generic vs guideline-directed prompt package conditions was prespecified in this study [<xref ref-type="bibr" rid="ref16">16</xref>-<xref ref-type="bibr" rid="ref18">18</xref>].</p>
      </sec>
      <sec>
        <title>Repetition and Data Collection</title>
        <p>Each scenario was run in 3 independent repetitions per model and per prompt arm, thereby enabling intramodel consistency to be evaluated as a separate end point. Each repetition was conducted in a new and independent session. The total number of observations was 1080 for the main analysis (60 scenarios × 3 models × 2 arms × 3 repetitions) and 108 for exploratory arm 3 (12 scenarios × 3 models × 3 repetitions).</p>
        <p>Previous medical benchmark studies have demonstrated that accuracy and stability do not always co-occur when identical inputs are repeatedly presented to the same model [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. Accordingly, the triple-repetition approach was defined as a core methodological component of this study.</p>
        <p>These parameters—model identity, interface, session independence, and configuration—were fixed and documented at the level of the study protocol. Query-level logs, including build identifiers or response timestamps for individual queries, were not retained. Data collection was conducted between March 25 and March 31, 2026.</p>
      </sec>
      <sec>
        <title>Evaluation Process</title>
        <p>Model outputs were evaluated independently by 2 obstetrics and gynecology specialists (ÖOE and CE) in accordance with the prespecified rubric. Responses were presented as coded output sets with model and prompt-arm identifiers removed. The evaluators were therefore blinded to model and prompt-arm identity, but they were not independent of gold-standard construction: both had participated in developing the scenarios and the answer key. Complete masking may also not have been achievable because of the characteristic response styles of different models. These 2 constraints are distinct and are addressed separately in the Strengths and Limitations section. Disagreements between the 2 evaluators were resolved by consensus. An illustrative scoring example demonstrating the 3-tier classification is provided in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p>
      </sec>
      <sec>
        <title>End Points and Error Classification</title>
        <p>Each model response was classified into 1 of 3 categories. <italic>Exact correct</italic> was defined as an initial management recommendation fully concordant with the ASCCP guidelines and clinically accurate. <italic>Safe but imperfect</italic> encompassed responses that did not violate the overall framework of clinical safety but contained incomplete phrasing, omission of acceptable alternatives, or minor detail errors. <italic>Unsafe major error</italic> denoted a management recommendation incorrect to a degree carrying clinically meaningful harm potential. This 3-tier classification was based on the principle that not all incorrect responses are clinically equivalent and that a safety-oriented distinction is necessary [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref14">14</xref>].</p>
        <p>Seven error-subtype options were available in the scoring workbook. Undermanagement was defined as omission or delay of indicated colposcopy or treatment escalation other than an error specific to the expedited-treatment threshold. Overmanagement was defined as recommendation of unnecessary procedures or colposcopy. Wrong expedited treatment was defined as an error specific to the expedited-treatment threshold, that is, recommending expedited treatment where the immediate risk falls below that threshold, or omitting it where the guideline states that expedited treatment is preferred. Wrong surveillance interval was defined as recommending a follow-up interval inconsistent with the guidelines. History neglect was defined as failure to incorporate prior screening history into risk estimation. Genotype misinterpretation was defined as failure to apply exception rules specific to HPV 16/18. Age-threshold error was defined as incorrect application of an age-dependent decision rule. Where an error could in principle satisfy more than 1 definition, the expedited-treatment threshold took precedence: an error was coded as wrong expedited treatment only if it concerned that threshold itself, and as undermanagement otherwise.</p>
        <p>The primary end point was the unsafe major error-free rate. The main secondary end point was the rate of exact concordance with the prespecified gold-standard management decision (exact concordance rate). Additional secondary end points included error subtype distribution, performance by scenario complexity, and intramodel consistency. The unsafe major error-free rate was designated as the confirmatory primary end point; all other analyses, although prespecified, were interpreted as exploratory.</p>
      </sec>
      <sec>
        <title>Statistical Analysis</title>
        <p>Descriptive proportions are reported with 95% CIs adjusted for within-scenario clustering. The design effect was calculated as 1 + (m − 1) × ICC, where m is the number of observations contributed by each scenario (m=3 for model-by-arm estimates, m=18 for overall complexity-stratum estimates, and m=6 for model-by-complexity estimates) and the intraclass correlation coefficient (ICC) was estimated by 1-way ANOVA separately for each outcome. Wilson intervals were then computed on the effective sample size. In 4 cells (1 model-by-arm cell and 3 model-by-complexity cells), all responses fell in the same category, so the ICC was undefined; for those cells, the number of independent scenarios was used as the effective sample size. A scenario-cluster bootstrap produced closely comparable intervals for the remaining cells. Unadjusted response-level Wilson intervals are provided for comparison in Table S3 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p>
        <p>Between-model comparisons were performed as omnibus and pairwise contrasts from cluster-aware generalized estimating equation (GEE) models, with Bonferroni correction applied across the 3 pairwise comparisons within each outcome and arm. The response-level McNemar test used in the original analysis did not account for the repeated-measures structure and has been removed from the primary inference; the prompt-arm effect is reported from the GEE model, with a complementary scenario-level paired analysis. For that complementary analysis, the 3 repetitions for each scenario, model, and arm were averaged; CIs were obtained by resampling scenarios with replacement, and 2-sided <italic>P</italic> values by exact sign-flip permutation enumeration where computationally feasible, with a Monte Carlo sign-flip test using 200,000 draws where enumeration was not feasible. Between-model heterogeneity in the scenario-level difference was tested by a cluster-robust Wald <italic>F</italic> test from an ordinary least squares model of the difference on model, with SEs clustered by scenario.</p>
        <p>GEE logistic regression was used to account for the repeated-measures structure [<xref ref-type="bibr" rid="ref19">19</xref>]. This approach was chosen to address within-cluster dependence arising from the evaluation of the same scenario across different model, prompt, and repetition combinations. The unsafe major error-free response (binary) and exact concordance (binary) were modeled separately as dependent variables. Fixed effects included model, prompt arm, and scenario complexity; the clustering variable was scenario number. An exchangeable correlation structure was used.</p>
        <p>Interrater agreement was calculated using Cohen κ coefficient (unweighted and linear weighted), with CIs for both coefficients derived from the corresponding asymptotic standard error [<xref ref-type="bibr" rid="ref20">20</xref>]. As a post hoc sensitivity analysis addressing adjudication asymmetry, the primary end point was recomputed separately using each evaluator’s independent ratings. A scenario-cluster bootstrap resampling the 60 scenarios with replacement (4000 replications) yielded 0.801-0.866 for the linearly weighted coefficient, reported alongside the asymptotic interval rather than in place of it. Because several strata contained no unsafe major errors, a Firth penalized logistic regression with profile penalized likelihood confidence intervals was fitted as a sensitivity analysis addressing separation; this model does not account for within-scenario clustering and does not replace the cluster-aware GEEs [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. Intramodel consistency was defined as the proportion of scenarios in which the same consensus classification was obtained across all 3 repetitions.</p>
        <p>Arm 3 was analyzed descriptively. Formal significance testing is not reported for this arm because the number of discordant scenarios is too small to support inference.</p>
        <p>Statistical analyses were performed using IBM SPSS Statistics (version 26.0; IBM Corp) and Python 3 with the <italic>statsmodels</italic>, <italic>scikit-learn</italic>, and <italic>SciPy</italic> libraries. Analysis code is provided in <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>. A <italic>P</italic> value of less than .05 was considered statistically significant.</p>
        <p>Sample size was determined by the systematic structure of the predefined scenario coverage matrix. As this study used a scenario-based comparative benchmark design rather than a conventional superiority or equivalence hypothesis test, a precision-based justification approach was adopted instead of a traditional a priori power analysis [<xref ref-type="bibr" rid="ref23">23</xref>]. With 180 observations per model per prompt arm, the half-width of the CI for proportions between 80% and 100% was projected to remain within approximately ±2.1 to ±5.9 percentage points (pp) before adjustment for within-scenario clustering.</p>
        <p>Reporting follows the Chatbot Assessment Reporting Tool (CHART) statement for chatbot health advice studies [<xref ref-type="bibr" rid="ref24">24</xref>] and the Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis–Large Language Models (TRIPOD-LLM) reporting guideline for studies using LLMs [<xref ref-type="bibr" rid="ref25">25</xref>]. A completed CHART checklist is provided in <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Dataset Characteristics</title>
        <p>A total of 1080 model responses (60 scenarios × 3 models × 2 prompt arms × 3 repetitions) were included in the main analysis. This yielded 360 observations per model and 180 observations per model per prompt arm. Of the 60 scenarios, 48 were classified as core scenarios and 12 as challenge scenarios. The complexity distribution comprised 16 low-complexity scenarios (288 observations), 21 medium-complexity scenarios (378 observations), and 23 high-complexity scenarios (414 observations). In addition, 108 observations (12 challenge scenarios × 3 models × 3 repetitions) were collected for exploratory arm 3. The overall characteristics of the study dataset are presented in <xref ref-type="table" rid="table1">Table 1</xref>.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Study design and dataset characteristics. Arm 3 was conducted on the 12 challenge scenarios using the arm 2 prompts rerun in separate sessions and is not part of the 1080 primary-analysis responses. The evaluators were masked to model and prompt-arm identity but had participated in construction of the gold standard, as described in the Strengths and Limitations section.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="0"/>
            <col width="300"/>
            <col width="700"/>
            <thead>
              <tr valign="top">
                <td colspan="2">Characteristic</td>
                <td>Value, n</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="2">Total observations (primary analysis)</td>
                <td>1080</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Models evaluated</td>
                <td>3 (GPT-5.3, Gemini 3 Flash, and DeepSeek V3.2)</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Prompt arms</td>
                <td>2 (baseline clinical prompt; guideline-directed prompt package)</td>
              </tr>
              <tr valign="top">
                <td colspan="3">Total scenarios (n=60)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Core scenarios</td>
                <td>48 (864 observations)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Challenge scenarios</td>
                <td>12 (216 observations)</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Repetitions per scenario (model and arm)</td>
                <td>3</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Observations per model</td>
                <td>360</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Observations per model per arm</td>
                <td>180</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Scenario complexity (low)</td>
                <td>16 scenarios (288 observations)</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Scenario complexity (medium)</td>
                <td>21 scenarios (378 observations)</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Scenario complexity (high)</td>
                <td>23 scenarios (414 observations)</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Additional exploratory arm 3 observations</td>
                <td>108</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Gold standard</td>
                <td>Prespecified gold-standard decisions based on the 2019 ASCCP<sup>a</sup> risk-based management consensus guidelines and supporting risk-estimate tables; post hoc source verification also reviewed official updates through 2023</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Evaluation</td>
                <td>Two obstetrics and gynecology specialists, scoring independently with model and prompt-arm identity masked</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Primary outcome</td>
                <td>Unsafe major error-free rate</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Main secondary outcome</td>
                <td>Exact concordance rate</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>ASCCP: American Society for Colposcopy and Cervical Pathology.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Primary End Point: Unsafe Major Error-Free Rate</title>
        <p>Unsafe major error-free rates by model and prompt arm are presented in <xref ref-type="table" rid="table2">Table 2</xref>. GPT-5.3 produced an unsafe major error-free response rate of 95.6% (95% CI 87.6%-98.5%) in the baseline clinical prompt arm (arm 1), which increased to 100% (95% CI 94%-100%) in the guideline-directed prompt package arm (arm 2). The corresponding rates for Gemini 3 Flash were 80% (95% CI 69%-87.8%) and 98.9% (95% CI 94%-99.8%), respectively, whereas those for DeepSeek V3.2 were 58.9% (95% CI 46.9%-69.9%) and 75% (95% CI 63.2%-84%), respectively. All CIs are adjusted for within-scenario clustering; unadjusted response-level CIs are provided in Table S3 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p>
        <p>In arm 1, the difference among the 3 models was statistically significant in a cluster-aware omnibus contrast (Wald <italic>χ</italic><sup>2</sup><sub>2</sub>=26.83; <italic>P</italic>&lt;.001). All 3 pairwise comparisons remained significant after Bonferroni correction (Table S4 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). In arm 2, GPT-5.3 produced no unsafe major errors, so the omnibus contrast and the 2 pairwise comparisons involving GPT-5.3 were not estimable for this outcome; the observed rates were 100%, 98.9%, and 75% for GPT-5.3, Gemini 3 Flash, and DeepSeek V3.2, respectively, and the contrast between Gemini 3 Flash and DeepSeek V3.2 remained significant (<italic>P</italic>=.001). The distribution of response classifications across all scenarios by model and prompt arm is visualized in <xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Response classification and performance by model and prompt arm. Each model-arm cell comprises 180 responses (60 scenarios × 3 repetitions). Arm 1 is the baseline clinical prompt; arm 2 is the guideline-directed prompt package. CIs are adjusted for within-scenario clustering using a design-effect correction: design effect = 1 + (m − 1) × ICC, m=3, with Wilson intervals computed on the effective sample size. In the GPT-5.3 arm 2 cell, all responses fell in the same category, so the ICC<sup>a</sup> was undefined and the number of independent scenarios (n=60) was used as the effective sample size. Unadjusted response-level Wilson intervals are provided in Table S3 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="120"/>
            <col width="50"/>
            <col width="120"/>
            <col width="160"/>
            <col width="160"/>
            <col width="190"/>
            <col width="200"/>
            <thead>
              <tr valign="top">
                <td>Model</td>
                <td>Arm</td>
                <td>Exact correct, n</td>
                <td>Safe but imperfect, n</td>
                <td>Unsafe major error, n</td>
                <td>Unsafe major error-free rate, % (95% CI)</td>
                <td>Exact concordance rate, % (95% CI)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>GPT-5.3</td>
                <td>1</td>
                <td>156</td>
                <td>16</td>
                <td>8</td>
                <td>95.6 (87.6-98.5)</td>
                <td>86.7 (75.8-93.1)</td>
              </tr>
              <tr valign="top">
                <td>GPT-5.3</td>
                <td>2</td>
                <td>177</td>
                <td>3</td>
                <td>0</td>
                <td>100 (94-100)</td>
                <td>98.3 (95.2-99.4)</td>
              </tr>
              <tr valign="top">
                <td>Gemini 3 Flash</td>
                <td>1</td>
                <td>104</td>
                <td>40</td>
                <td>36</td>
                <td>80 (69-87.8)</td>
                <td>57.8 (45.3-69.4)</td>
              </tr>
              <tr valign="top">
                <td>Gemini 3 Flash</td>
                <td>2</td>
                <td>169</td>
                <td>9</td>
                <td>2</td>
                <td>98.9 (94-99.8)</td>
                <td>93.9 (85.6-97.5)</td>
              </tr>
              <tr valign="top">
                <td>DeepSeek V3.2</td>
                <td>1</td>
                <td>68</td>
                <td>38</td>
                <td>74</td>
                <td>58.9 (46.9-69.9)</td>
                <td>37.8 (26.7-50.3)</td>
              </tr>
              <tr valign="top">
                <td>DeepSeek V3.2</td>
                <td>2</td>
                <td>98</td>
                <td>37</td>
                <td>45</td>
                <td>75 (63.2-84)</td>
                <td>54.4 (42-66.3)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup>ICC: intraclass correlation coefficient.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Performance by Prompt Condition</title>
        <p>The guideline-directed prompt package (arm 2) was associated with a higher unsafe major error-free rate in all 3 models compared with the baseline clinical prompt (arm 1). In the GEE model, the odds of an unsafe major error-free response were 3.76-fold higher in arm 2 (95% CI 2.45-5.77; <italic>P</italic>&lt;.001). In a complementary scenario-level analysis, the absolute improvement was +4.4 pp for GPT-5.3 (95% CI 0.0-10.0 pp; <italic>P</italic>=.25), +18.9 pp for Gemini 3 Flash (95% CI 10.0-28.3 pp; <italic>P</italic>&lt;.001), and +16.1 pp for DeepSeek V3.2 (95% CI 7.2-26.7 pp; <italic>P</italic>=.003). Improvement was observed in 3, 14, and 13 scenarios, respectively, with 4 scenarios showing deterioration for DeepSeek V3.2 and none for the other 2 models. For GPT-5.3, only 3 scenarios differed between arms, so the smallest attainable permutation <italic>P</italic> value is .25 and the reported value lies at that floor.</p>
      </sec>
      <sec>
        <title>Exact Concordance With ASCCP Recommendations</title>
        <p>Regarding the main secondary end point, GPT-5.3 produced responses fully concordant with the ASCCP guidelines in 86.7% of cases (95% CI 75.8%-93.1%) in arm 1 and 98.3% of cases (95% CI 95.2%-99.4%) in arm 2. The corresponding rates for Gemini 3 Flash were 57.8% (95% CI 45.3%-69.4%) and 93.9% (95% CI 85.6%-97.5%), respectively, whereas those for DeepSeek V3.2 were 37.8% (95% CI 26.7%-50.3%) and 54.4% (95% CI 42%-66.3%), respectively. In both arms, the overall difference among the 3 models was statistically significant in cluster-aware omnibus contrasts (arm 1: Wald <italic>χ</italic><sup>2</sup><sub>2</sub>=38.31; <italic>P</italic>&lt;.001 and arm 2: Wald <italic>χ</italic><sup>2</sup><sub>2</sub>=44.83; <italic>P</italic>&lt;.001). In arm 2, the pairwise contrast between GPT-5.3 and Gemini 3 Flash did not meet the Bonferroni-corrected threshold (<italic>P</italic>=.04; threshold=.0167).</p>
      </sec>
      <sec>
        <title>GEE Logistic Regression Analysis</title>
        <p>The results of the GEE logistic regression analysis, performed to account for the repeated-measures structure, are presented in <xref ref-type="table" rid="table3">Table 3</xref>. For the unsafe major error-free rate, taking GPT-5.3 as the reference, Gemini 3 Flash showed a significantly lower likelihood of producing a safe response (odds ratio [OR] 0.27; 95% CI 0.16-0.45; <italic>P</italic>&lt;.001). This likelihood decreased even more markedly for DeepSeek V3.2 (OR 0.05; 95% CI 0.02-0.12; <italic>P</italic>&lt;.001). The guideline-directed prompt package increased the likelihood of an unsafe major error-free response approximately 3.8-fold (OR 3.76; 95% CI 2.45-5.77; <italic>P</italic>&lt;.001). In high-complexity scenarios, the likelihood of an unsafe major error-free response was significantly lower than in low-complexity scenarios (OR 0.05; 95% CI 0.01-0.24; <italic>P</italic>&lt;.001), whereas the difference for medium complexity did not reach statistical significance (OR 0.30; 95% CI 0.07-1.37; <italic>P</italic>=.12).</p>
        <p>A similar pattern was observed for exact concordance, although effect sizes were more pronounced. The guideline-directed prompt package increased the likelihood of exact concordance 8.3-fold (OR 8.27; 95% CI 4.14-16.52; <italic>P</italic>&lt;.001). Compared with GPT-5.3, DeepSeek V3.2 had a substantially lower likelihood of exact concordance (OR 0.014; 95% CI 0.004-0.045; <italic>P</italic>&lt;.001). High complexity was associated with a marked reduction in the likelihood of exact concordance compared with low complexity (OR 0.005; 95% CI 0.001-0.032; <italic>P</italic>&lt;.001). A model-by-prompt interaction was examined for exact concordance and was not statistically significant on the odds scale (DeepSeek V3.2 × arm 2: OR 0.33, 95% CI 0.08-1.42; <italic>P</italic>=.14 and Gemini 3 Flash × arm 2: OR 2.52, 95% CI 0.47-13.48; <italic>P</italic>=.28). The corresponding interaction model for the unsafe major error-free outcome was not estimable because 9 of the 18 model-by-arm-by-complexity cells contained 0 unsafe major errors. A Firth penalized sensitivity analysis confirmed that the direction of the prompt-package effect was unchanged, whereas the very wide profile penalized likelihood intervals confirmed substantial effect-size instability associated with sparse and 0-event cells (Table S5 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). In the complementary scenario-level analysis, between-model heterogeneity in the absolute improvement was statistically significant for both outcomes (unsafe major error-free: <italic>F</italic><sub>2,59</sub>=6.97; <italic>P</italic>=.002 and exact concordance: <italic>F</italic><sub>2,59</sub>=6.38; <italic>P</italic>=.003).</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>GEE<sup>a</sup> models for the primary and main secondary outcomes. Binomial GEEs with a logit link, scenario as the clustering unit, an exchangeable working correlation structure, and robust SEs. Both primary main-effects models converged without warnings. Estimates involving strata with no observed unsafe major errors should be interpreted with caution. Nine of the 18 model-by-arm-by-complexity cells contained no unsafe major errors: GPT-5.3 in the low- and medium-complexity strata of arm 1 and in all 3 strata of arm 2, Gemini 3 Flash in the low-complexity stratum of arm 1 and in the low- and medium-complexity strata of arm 2, and DeepSeek V3.2 in the low-complexity stratum of arm 2. A model-by-arm interaction was examined for exact concordance and was not statistically significant (Table S5 in Multimedia Appendix 5); the corresponding interaction model for the unsafe major error-free outcome was not estimable because 9 of 18 model-by-arm-by-complexity cells contained 0 unsafe major errors.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="270"/>
            <col width="200"/>
            <col width="500"/>
            <thead>
              <tr valign="top">
                <td colspan="2">Outcome and term</td>
                <td>OR<sup>b</sup> (95% CI)</td>
                <td><italic>P</italic> value</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="4">Unsafe major error-free</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Gemini 3 Flash vs GPT-5.3</td>
                <td>0.269 (0.162-0.447)</td>
                <td>&lt;.001</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>DeepSeek V3.2 vs GPT-5.3</td>
                <td>0.051 (0.022-0.119)</td>
                <td>&lt;.001</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Arm 2 vs arm 1</td>
                <td>3.756 (2.446-5.767)</td>
                <td>&lt;.001</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Medium vs low complexity</td>
                <td>0.303 (0.067-1.369)</td>
                <td>.12</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>High vs low complexity</td>
                <td>0.047 (0.009-0.242)</td>
                <td>&lt;.001</td>
              </tr>
              <tr valign="top">
                <td colspan="4">Exact concordance</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Gemini 3 Flash vs GPT-5.3</td>
                <td>0.145 (0.073-0.289)</td>
                <td>&lt;.001</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>DeepSeek V3.2 vs GPT-5.3</td>
                <td>0.014 (0.004-0.045)</td>
                <td>&lt;.001</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Arm 2 vs arm 1</td>
                <td>8.269 (4.140-16.518)</td>
                <td>&lt;.001</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Medium vs low complexity</td>
                <td>0.056 (0.010-0.329)</td>
                <td>.001</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>High vs low complexity</td>
                <td>0.005 (0.001-0.032)</td>
                <td>&lt;.001</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table3fn1">
              <p><sup>a</sup>GEE: generalized estimating equation.</p>
            </fn>
            <fn id="table3fn2">
              <p><sup>b</sup>OR: odds ratio.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Performance by Scenario Complexity</title>
        <p>The unsafe major error rate increased markedly with increasing scenario complexity (<xref ref-type="table" rid="table4">Table 4</xref>). The unsafe major error rate was 3.1% (95% CI 1.1%-8.3%) in low-complexity scenarios, 9.5% (95% CI 5.3%-16.5%) in medium-complexity scenarios, and 29% (95% CI 19.9%-40.1%) in high-complexity scenarios. The corresponding gradient is reflected in the GEE model, in which high complexity was associated with markedly lower odds of an unsafe major error-free response compared with low complexity (<xref ref-type="table" rid="table3">Table 3</xref>).</p>
        <p>When examined by model (<xref ref-type="table" rid="table5">Table 5</xref>), the most striking findings emerged in the high-complexity group. In high-complexity scenarios, GPT-5.3 produced unsafe major error-free responses in 94.2% of cases, whereas the corresponding rates were 76.8% for Gemini 3 Flash and only 42% for DeepSeek V3.2. In low-complexity scenarios, by contrast, both GPT-5.3 and Gemini 3 Flash achieved a 100% unsafe major error-free rate, whereas DeepSeek V3.2 still produced unsafe major errors in 9.4% of cases even in this group (<xref ref-type="table" rid="table5">Table 5</xref>).</p>
        <table-wrap position="float" id="table4">
          <label>Table 4</label>
          <caption>
            <p>Response classification and performance by scenario complexity. Complexity strata pool observations across models and prompt arms, so each scenario contributes 18 observations (3 models × 2 arms × 3 repetitions); the design-effect correction therefore uses m=18 for these estimates, and ICCs<sup>a</sup> were estimated separately for each outcome. The corresponding unsafe major error rates were 3.1% (95% CI 1.1%-8.3%) for low complexity, 9.5% (95% CI 5.3%-16.5%) for medium complexity, and 29% (95% CI 19.9%-40.1%) for high complexity.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="100"/>
            <col width="120"/>
            <col width="120"/>
            <col width="150"/>
            <col width="150"/>
            <col width="180"/>
            <col width="180"/>
            <thead>
              <tr valign="top">
                <td>Complexity</td>
                <td>Observations, n</td>
                <td>Exact correct, n</td>
                <td>Safe but imperfect, n</td>
                <td>Unsafe major error, n</td>
                <td>Unsafe major error-free rate, % (95% CI)</td>
                <td>Exact concordance rate, % (95% CI)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Low</td>
                <td>288</td>
                <td>275</td>
                <td>4</td>
                <td>9</td>
                <td>96.9 (91.7-98.9)</td>
                <td>95.5 (90.1-98)</td>
              </tr>
              <tr valign="top">
                <td>Medium</td>
                <td>378</td>
                <td>297</td>
                <td>45</td>
                <td>36</td>
                <td>90.5 (83.5-94.7)</td>
                <td>78.6 (69.1-85.8)</td>
              </tr>
              <tr valign="top">
                <td>High</td>
                <td>414</td>
                <td>200</td>
                <td>94</td>
                <td>120</td>
                <td>71 (59.9-80.1)</td>
                <td>48.3 (39.5-57.3)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table4fn1">
              <p><sup>a</sup>ICC: intraclass correlation coefficient.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <table-wrap position="float" id="table5">
          <label>Table 5</label>
          <caption>
            <p>Performance by model and complexity stratum. Each scenario contributes 6 observations (2 arms × 3 repetitions), so m=6 for those estimates. In 3 cells, all responses fell in the same category and the ICC<sup>a</sup> was undefined; for those cells, the number of independent scenarios was used as the effective sample size, as in <xref ref-type="table" rid="table2">Table 2</xref>. This panel is descriptive and was not used for formal inference.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="8"/>
            <col width="242"/>
            <col width="0"/>
            <col width="170"/>
            <col width="0"/>
            <col width="258"/>
            <col width="322"/>
            <thead>
              <tr valign="top">
                <td colspan="3">Model and complexity</td>
                <td colspan="2">Observations, n</td>
                <td>Unsafe major errors, n (%)</td>
                <td>Unsafe major error-free rate, % (95% CI)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="7">GPT-5.3</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Low</td>
                <td colspan="2">96</td>
                <td colspan="2">0 (0)</td>
                <td>100 (80.6-100)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Medium</td>
                <td colspan="2">126</td>
                <td colspan="2">0 (0)</td>
                <td>100 (84.5-100)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>High</td>
                <td colspan="2">138</td>
                <td colspan="2">8 (5.8)</td>
                <td>94.2 (84.4-98)</td>
              </tr>
              <tr valign="top">
                <td colspan="7">Gemini 3 Flash</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Low</td>
                <td colspan="2">96</td>
                <td colspan="2">0 (0)</td>
                <td>100 (80.6-100)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Medium</td>
                <td colspan="2">126</td>
                <td colspan="2">6 (4.8)</td>
                <td>95.2 (86.3-98.5)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>High</td>
                <td colspan="2">138</td>
                <td colspan="2">32 (23.2)</td>
                <td>76.8 (64.7-85.7)</td>
              </tr>
              <tr valign="top">
                <td colspan="7">DeepSeek V3.2</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Low</td>
                <td colspan="2">96</td>
                <td colspan="2">9 (9.4)</td>
                <td>90.6 (76.5-96.6)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Medium</td>
                <td colspan="2">126</td>
                <td colspan="2">30 (23.8)</td>
                <td>76.2 (61.7-86.4)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>High</td>
                <td colspan="2">138</td>
                <td colspan="2">80 (58)</td>
                <td>42 (25.8-60.2)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table5fn1">
              <p><sup>a</sup>ICC: intraclass correlation coefficient.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Unsafe Major Error Subtypes</title>
        <p>A total of 165 unsafe major errors were identified. The most frequently observed error subtype was undermanagement, accounting for 39.4% (n=65) of all errors. This was followed by genotype misinterpretation (n=48, 29.1%), history neglect (n=37, 22.4%), and overmanagement (n=15, 9.1%; <xref ref-type="table" rid="table6">Table 6</xref>).</p>
        <p>The distribution of error subtypes differed markedly across models. All 8 errors produced by GPT-5.3 were classified as history neglect; this model did not produce any genotype misinterpretation, undermanagement, or overmanagement errors. Among the 38 errors generated by Gemini 3 Flash, undermanagement (n=14), genotype misinterpretation (n=13), and history neglect (n=11) were observed at similar frequencies. DeepSeek V3.2 had the highest overall error burden with 119 errors, and its error profile was dominated by undermanagement (n=51), followed by genotype misinterpretation (n=35), history neglect (n=18), and overmanagement (n=15). Overmanagement errors were observed exclusively in DeepSeek V3.2.</p>
        <p>When evaluated by prompt arm, 71.5% (n=118) of all errors occurred in arm 1 and 28.5% (n=47) in arm 2, supporting the conclusion that the guideline-directed prompt package reduced the overall error burden (<xref ref-type="table" rid="table6">Table 6</xref>).</p>
        <table-wrap position="float" id="table6">
          <label>Table 6</label>
          <caption>
            <p>Distribution of unsafe major error subtypes. Seven error-subtype options were available in the scoring workbook. Each unsafe major error was assigned 1 primary subtype by consensus of the 2 evaluators. Wrong surveillance interval, age-threshold error, and wrong expedited treatment were available as options but received no consensus assignments. Because subtype assignment was made by consensus without an independent reliability assessment, these zero counts should not be interpreted as evidence that the corresponding error types cannot occur.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="200"/>
            <col width="170"/>
            <col width="120"/>
            <col width="150"/>
            <col width="150"/>
            <col width="210"/>
            <thead>
              <tr valign="top">
                <td>Error subtype</td>
                <td>Total, n (%)</td>
                <td>GPT-5.3, n</td>
                <td>Gemini 3 Flash, n</td>
                <td>DeepSeek V3.2, n</td>
                <td>Arm 1/arm 2, n</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Undermanagement</td>
                <td>65 (39.4)</td>
                <td>0</td>
                <td>14</td>
                <td>51</td>
                <td>48/17</td>
              </tr>
              <tr valign="top">
                <td>Genotype misinterpretation</td>
                <td>48 (29.1)</td>
                <td>0</td>
                <td>13</td>
                <td>35</td>
                <td>29/19</td>
              </tr>
              <tr valign="top">
                <td>History neglect</td>
                <td>37 (22.4)</td>
                <td>8</td>
                <td>11</td>
                <td>18</td>
                <td>26/11</td>
              </tr>
              <tr valign="top">
                <td>Overmanagement</td>
                <td>15 (9.1)</td>
                <td>0</td>
                <td>0</td>
                <td>15</td>
                <td>15/0</td>
              </tr>
              <tr valign="top">
                <td>Wrong surveillance interval</td>
                <td>0 (0)</td>
                <td>0</td>
                <td>0</td>
                <td>0</td>
                <td>0/0</td>
              </tr>
              <tr valign="top">
                <td>Age-threshold error</td>
                <td>0 (0)</td>
                <td>0</td>
                <td>0</td>
                <td>0</td>
                <td>0/0</td>
              </tr>
              <tr valign="top">
                <td>Wrong expedited treatment</td>
                <td>0 (0)</td>
                <td>0</td>
                <td>0</td>
                <td>0</td>
                <td>0/0</td>
              </tr>
              <tr valign="top">
                <td>Total</td>
                <td>165 (100)</td>
                <td>8</td>
                <td>38</td>
                <td>119</td>
                <td>118/47</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
      <sec>
        <title>Interrater Agreement</title>
        <p>The raw agreement rate between the 2 evaluators, who scored responses independently, was 88.9% (960/1080). Cohen κ coefficient was 0.770 (95% CI 0.732-0.807) in the unweighted analysis and 0.839 (95% CI 0.811-0.867) in the linearly weighted analysis, corresponding to substantial and almost perfect agreement, respectively, under the benchmarks of Landis and Koch [<xref ref-type="bibr" rid="ref20">20</xref>] (Table S1 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>).</p>
        <p>Most disagreements occurred between the exact correct and safe but imperfect categories. A total of 81 responses were classified as exact correct by the first evaluator and safe but imperfect by the second, whereas 11 responses showed the opposite pattern. Cross-category disagreements between safe but imperfect and unsafe major error were less frequent, with 13 responses classified in the former-to-latter direction and 15 in the latter-to-former direction.</p>
      </sec>
      <sec>
        <title>Intramodel Consistency</title>
        <p>Intramodel consistency, defined as the proportion of scenarios in which the same consensus classification was obtained across all 3 repetitions, varied across models and prompt arms. GPT-5.3 demonstrated complete consistency in 98.3% (59/60) of scenarios in arm 1 and in 95% (57/60) of scenarios in arm 2. The corresponding rates for Gemini 3 Flash were 90% (54/60) and 95% (57/60), respectively, whereas those for DeepSeek V3.2 were 86.7% (52/60) and 93.3% (56/60), respectively. No model-prompt arm combination produced any scenario in which all 3 repetitions received 3 different classifications (<xref ref-type="supplementary-material" rid="app9">Multimedia Appendix 9</xref>).</p>
      </sec>
      <sec>
        <title>Exploratory Arm 3: Checklist-Based Self-Verification</title>
        <p>A total of 108 observations were evaluated under exploratory arm 3 (Table S2 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). Because GPT-5.3 had already produced no unsafe major errors in arm 2, no self-correction was observed. For Gemini 3 Flash, postchecklist correction was identified in 8.3% (3/36) of responses, and for DeepSeek V3.2 in 16.7% (6/36) of responses. Formal significance testing is not reported for this arm because the number of discordant scenarios is too small to support inference. No model showed a reverse transition in which a previously safe response became classified as an unsafe major error after checklist-based review.</p>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Results</title>
        <p>In this scenario-based benchmark study, the performance of LLMs in the 2019 ASCCP risk-based initial management framework was evaluated not only in terms of accuracy but also from the perspective of clinical safety. The principal results of the study can be summarized under 4 main themes. First, a clear performance hierarchy was observed among the 3 models, with GPT-5.3 demonstrating the strongest performance in terms of both the unsafe major error-free rate and exact guideline concordance. Second, the guideline-directed prompt package was associated with improved performance across all models, with larger absolute gains in the models with weaker baseline performance, although the formal interaction test was not statistically significant on the odds scale. Third, the unsafe major error rate increased markedly with increasing scenario complexity, and this deterioration became especially striking at history-dependent decision nodes. Fourth, unsafe major errors were not randomly distributed but clustered around undermanagement, genotype misinterpretation, and history neglect. This pattern suggests that LLMs in this domain are challenged not only by knowledge limitations but also by difficulty in integrating risk context.</p>
        <p>One of the most important contributions of this study is that it moved beyond the conventional “correct/incorrect” dichotomy by evaluating performance using a safety-centered primary end point, namely the unsafe major error-free rate. Contemporary medical LLM benchmark literature emphasizes that not all errors are clinically equivalent, particularly in open-ended clinical tasks [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. In the safety-and-effectiveness benchmark study by Wang et al [<xref ref-type="bibr" rid="ref10">10</xref>], performance was reported to decline by an average of 13.3% in high-risk clinical scenarios. Our findings confirm this general observation in the context of cervical screening; more importantly, however, they demonstrate concretely at which decision nodes and through which error patterns this decline emerges. The study thus supports the view that, in medical LLM evaluation, the capacity to avoid generating harmful mismanagement recommendations is a clinically more meaningful metric than average accuracy alone.</p>
        <p>The most striking finding of this study was that GPT-5.3 produced no unsafe major errors under the guideline-directed prompt package and achieved an exact concordance rate of 98.3% within this benchmark. This rate refers to performance on a purposively constructed scenario set that deliberately oversamples complex and history-dependent decision nodes, and it is not a population-level estimate of clinical safety. Gemini 3 Flash reached a safety rate of 98.9% with the guideline-directed prompt package, thereby approaching GPT-5.3, but declined to 80% under baseline conditions. This finding indicates that Gemini 3 Flash reached a high unsafe major error-free rate within this benchmark when appropriately guided, but that this capacity is strongly dependent on prompt configuration. DeepSeek V3.2 exhibited the lowest performance in both arms and failed to exceed a safety rate of 75% even under the guideline-directed prompt package. This result demonstrates that while the guideline-directed prompt package improves performance, it is insufficient on its own when model-specific performance limitations are substantial.</p>
        <p>The effect of prompt strategy is among the more directly actionable findings of this study. In the GEE analysis, the guideline-directed prompt package increased the likelihood of an unsafe major error-free response 3.8-fold (OR 3.76, 95% CI 2.45-5.77) and the likelihood of exact concordance 8.3-fold (OR 8.27, 95% CI 4.14-16.52). The guideline-directed prompt package was therefore associated with substantially improved benchmark performance, after adjustment for model and scenario complexity. Because the package varied 3 components jointly—naming the guideline, directing risk-based reasoning, and requesting a structured output—the design cannot attribute this association to output structure alone. Kuerbanjiang et al [<xref ref-type="bibr" rid="ref16">16</xref>] likewise observed that prompting could improve performance in cervical cancer management; however, that effect was reported at a descriptive level rather than through a controlled comparison. Our study quantified the prompt-package effect in a cluster-aware GEE model and in a complementary scenario-level paired analysis. Absolute pp improvements differed across models in that complementary analysis; however, the formal interaction test for exact concordance was not statistically significant on the odds scale, and the corresponding interaction model for the safety outcome was affected by separation. The increase in the unsafe major error-free rate from 80% to 98.9% in Gemini 3 Flash is consistent with the prompt package constraining the response toward the guideline framework, although the mechanism was not directly measured. These findings support further evaluation of standardized, guideline-directed prompt templates in clinical decision support applications of LLMs.</p>
        <p>Scenario complexity showed the largest association with the unsafe major error rate in this study. The overall unsafe major error rate was only 3.1% in low-complexity scenarios but rose to 29% in high-complexity scenarios. In the cluster-aware analysis, high complexity was associated with a marked reduction in the likelihood of an unsafe major error-free response compared with low complexity (OR 0.05; 95% CI 0.01-0.24), and with a still larger reduction in the likelihood of exact concordance (OR 0.005; 95% CI 0.001-0.032). The difference between medium and low complexity did not reach significance for the safety outcome after adjustment for within-scenario clustering (OR 0.30; 95% CI 0.07-1.37), so the gradient is driven principally by the high-complexity stratum. These findings are consistent with those of Wang et al [<xref ref-type="bibr" rid="ref10">10</xref>], who demonstrated decreased LLM performance in high-risk scenarios, while revealing a more pronounced complexity-performance gradient specific to cervical screening management. This observation is also congruent with the inherent nature of the ASCCP risk-based framework: because the same current result may require entirely different initial management depending on prior history, correct management depends not on memorizing test-result patterns but on dynamically integrating risk context [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. In other words, the core vulnerability of LLMs in this domain may reflect not merely a lack of knowledge but a deficiency in risk-based contextual reasoning. Clinically, this finding is of critical importance, as undermanagement of a high-risk patient may lead to serious diagnostic delay, whereas overmanagement may result in unnecessary invasive procedures. Accordingly, if LLMs are to be considered for clinical decision support, the definition of complexity-stratified safety thresholds appears warranted.</p>
        <p>Error subtype analysis represents one of the most original contributions of this study to the existing literature. Whereas previous studies largely evaluated errors within a simple correct/incorrect framework [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref18">18</xref>], the present study classified unsafe major errors into clinically meaningful subtypes and demonstrated that different models exhibit distinct error profiles. The most frequently observed error type was undermanagement (39.4%), suggesting that the overall tendency of LLMs is not toward excessive caution but toward insufficient management. Genotype misinterpretation (29.1%) was identified as the second most common subtype, indicating that the models did not adequately internalize the exception rules specific to HPV 16/18. History neglect (22.4%) accounted for all 8 errors produced by GPT-5.3, revealing that even the best-performing model may at times fail to integrate prior screening history into risk estimation. Notably, overmanagement errors were observed exclusively in DeepSeek V3.2, making it the only model to produce both undermanagement and overmanagement errors. Taken together, these 3 dominant error clusters suggest that the central challenge of LLM performance in cervical screening management lies not merely in recognizing test labels but in integrating past and present information within a coherent risk-based reasoning framework. Risk-based and history-dependent domains such as ASCCP therefore represent particularly informative yet high-risk testing environments for LLMs. These differentiated error profiles provide valuable information for the future development of model-specific improvement strategies.</p>
        <p>Our intramodel consistency findings are consistent with recent multirun studies. Stelling et al [<xref ref-type="bibr" rid="ref13">13</xref>] reported that LLM performance on nuclear medicine board examinations varied across repetitions, and Güler et al [<xref ref-type="bibr" rid="ref14">14</xref>] described similar inconsistency patterns in hand fracture diagnosis. In our study, GPT-5.3 demonstrated complete consistency in 95% to 98.3% of scenarios, whereas DeepSeek V3.2 remained in the range of 86.7% to 93.3%. These findings support the view that accuracy and consistency should be evaluated as distinct constructs, and that reliability as a clinical decision support tool should depend not solely on single-run accuracy but also on reproducible performance.</p>
        <p>The exploratory arm 3 findings suggest that checklist-based self-verification demonstrates limited efficacy in its current form. Correction was observed in 16.7% of DeepSeek V3.2 responses and in 8.3% of Gemini 3 Flash responses; formal significance testing was not performed for this arm because the number of discordant scenarios is too small to support inference. More importantly, no model demonstrated a reverse transition in which a previously safe response deteriorated to the unsafe category after checklist review. This asymmetric finding suggests that self-verification mechanisms did not produce deterioration in this exploratory arm but do not yet function as a dependable safety net in their present form. Nevertheless, because this arm was applied to only 12 challenge scenarios for exploratory purposes, the results should be interpreted with caution.</p>
      </sec>
      <sec>
        <title>Comparison With Prior Work</title>
        <p>When compared with the existing LLM literature in the cervical cancer domain, our findings diverge at several important points. Pavone et al [<xref ref-type="bibr" rid="ref18">18</xref>] tested ChatGPT 4.0, DeepSeek R1, and Gemini 2.0 against ESGO guidelines and reported that all models demonstrated suboptimal accuracy; the effect of prompting was not evaluated in a controlled manner in that study. In our benchmark, Gemini 3 Flash achieved a 98.9% unsafe major error-free rate under the guideline-directed prompt package, although its exact concordance rate in the same condition was 93.9%. Direct numerical comparison with Pavone et al [<xref ref-type="bibr" rid="ref18">18</xref>] should be interpreted cautiously, because their highest quality score denotes complete correctness and therefore corresponds more closely to our exact concordance tier than to the absence of unsafe major errors. Our study nonetheless presents a more differentiated picture, in that high benchmark performance was attainable for some models under a guideline-directed prompt package. Yurtcu et al [<xref ref-type="bibr" rid="ref15">15</xref>] reported that ChatGPT showed acceptable performance on general knowledge questions but reduced accuracy on guideline-based questions; our findings support this observation and further quantify the performance decline with increasing complexity. The favorable usability findings of Angyal et al [<xref ref-type="bibr" rid="ref17">17</xref>] regarding a customized GPT model for patient education provide an additional perspective suggesting that informational and decision-support applications of LLMs carry different safety requirements. The findings of Ong et al [<xref ref-type="bibr" rid="ref12">12</xref>] that an LLM improved the accuracy of medication chart review when used alongside pharmacists are aligned with the central message of our study: LLMs should be positioned not as autonomous tools but as structured decision support systems under clinician supervision.</p>
        <p>A further question is what a generative model adds over a deterministic risk calculator for this specific task. The ASCCP framework is itself a rule-based algorithm, and computable versions of it are already being developed: the Centers for Disease Control and Prevention has pursued a multiyear initiative to translate narrative cervical screening guidelines into computable form for integration into electronic health record systems, explicitly motivated by the observation that the complexity and frequent updating of current evidence-based guidelines make it difficult for clinicians to keep pace [<xref ref-type="bibr" rid="ref26">26</xref>]. A deterministic calculator is reproducible by construction and cannot drift between runs, and for the final risk computation it is likely to be more dependable than a generative model. The potential contribution of an LLM lies elsewhere: extracting the relevant parameters from unstructured clinical narrative, operating where no structured decision-support infrastructure exists, and explaining the basis of a recommendation in natural language. Our findings indicate that this flexibility carries a safety cost precisely at the history-dependent decision nodes where the extraction task is hardest, which argues for pairing rather than substitution.</p>
      </sec>
      <sec>
        <title>Strengths and Limitations</title>
        <p>Several strengths of this study should be noted. First, the evaluation focused on a single risk-based management system—the ASCCP framework—providing a more defensible gold standard compared with broad question sets combining heterogeneous guidelines [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref6">6</xref>]. Second, the study used an open-ended, scenario-based design aimed at overcoming the limitations of multiple-choice examination formats [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Third, the safety-centered error classification allowed the separation of accuracy from harm potential. Fourth, the use of 3 independent repetitions enabled the assessment of intramodel consistency, an approach methodologically aligned with recent studies demonstrating that accuracy and stability are not necessarily equivalent [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. Finally, 2 specialist evaluators scored the outputs independently, and the high κ values support internal consistency of the scoring process.</p>
        <p>Several limitations should also be acknowledged. First, the 2 evaluators were not independent of gold-standard construction: both had participated in developing the scenarios and the answer key, and individuals who author an answer key and then apply it are structurally exposed to confirmation bias, particularly at borderline decisions. The interrater κ demonstrates internal consistency of the rating team rather than independence from the constructed gold standard. Disagreement was asymmetric rather than random, and the consensus classification coincided with the first evaluator’s independent score in 116 of the 120 discordant observations. In a post hoc sensitivity analysis, the primary end point was recomputed separately from each evaluator’s independent scores; rates differed by at most 2.8 pp, and neither the model hierarchy nor the direction of the prompt-package effect changed. As a further check, all 60 prespecified gold-standard decisions were verified post hoc against the cited ASCCP guidance; this is source verification of the answer key and not an independent external validation.</p>
        <p>Second, the verbatim model outputs were not prospectively archived as part of the study dataset. Because queries were conducted in independent, non–account-linked web sessions, no recoverable session history was subsequently available. Consequently, the mapping from response text to classification category cannot now be independently readjudicated. The coded observation-level dataset (<xref ref-type="supplementary-material" rid="app10">Multimedia Appendix 10</xref>), the gold-standard rubric with its source references (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), the prompt templates (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>), and the analysis code (<xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>) are provided, but this does not substitute for the original response texts, and this is the principal transparency limitation of the study.</p>
        <p>Third, no query-level version identifier, API request identifier, or pinned model checkpoint was recorded, so the exact server-side state of the tested systems cannot be verified or reproduced by another team. This is distinct from, and more consequential than, the general observation that commercial models evolve over time: the latter limits generalization to later versions, whereas the former limits verification of what was actually tested. All models were queried in their free, default, nonreasoning consumer configuration, and the reported rates should not be generalized to reasoning-enabled, retrieval-augmented, or enterprise-configured deployments of the same model families, which may perform differently.</p>
        <p>Fourth, scenario complexity and error subtype were each assigned once by investigator consensus, without an independent reliability assessment. No formal reliability coefficient can be computed for the complexity classification because independent ratings were not recorded. Minor and major deviation definitions were applied by consensus during evaluation rather than formalized in a separate written adjudication document. Fifth, synthetic scenarios may not capture all nuances of real-world clinical practice; however, this approach enabled controlled comparison under standardized conditions. Sixth, the scenario set deliberately oversamples complex and history-dependent decision nodes and is not a population-representative case mix, so the reported rates should be read as benchmark performance rather than as population-level safety estimates. Seventh, the study focused exclusively on initial management decisions and did not encompass other components of the ASCCP ecosystem, such as postcolposcopy surveillance and posttreatment follow-up. Eighth, ASCCP is a United States-specific guideline and the international generalizability of the findings is limited; however, the core principles of risk-based management are increasingly being adopted globally. Ninth, the study compared only LLMs and did not include a direct human clinician comparator arm for the same scenarios; although clinicians have previously been shown to experience difficulties with risk-based cervical screening management [<xref ref-type="bibr" rid="ref7">7</xref>], the absence of a direct human-model comparison represents a deliberate limitation of this study’s scope. Finally, while the 60 scenarios systematically covered the key decision nodes of the ASCCP framework, they do not represent all possible clinical combinations.</p>
        <p>Despite these limitations, the study delivers important practical messages regarding LLM performance in cervical screening management. Even in the best-case scenario, safety levels are sensitive to both model selection and prompt design; in the worst-case scenario, history-dependent and high-complexity decision nodes can generate clinically meaningful risk. Future research should therefore focus on 3 directions: first, broader ASCCP coverage encompassing postcolposcopy and posttreatment follow-up domains; second, human-AI interactive workflow studies involving real clinical users; and third, controlled evaluation of self-verification, checklist-based, and retrieval-augmented approaches. Such studies will help clarify the extent to which LLMs can be transformed into safe and useful assistive tools in cervical screening management.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>This scenario-based benchmark study demonstrated that the guideline concordance and safety-oriented benchmark performance of LLMs in the initial management of abnormal cervical screening results according to the 2019 ASCCP risk-based approach vary substantially by model, prompt strategy, and scenario complexity. GPT-5.3 demonstrated the highest benchmark safety and exact concordance performance, particularly under the guideline-directed prompt package; Gemini 3 Flash showed meaningful improvement but retained a prompt-dependent performance profile; and DeepSeek V3.2 showed the lowest benchmark performance under both prompt conditions.</p>
        <p>The guideline-directed prompt package was associated with improved performance on both the unsafe major error-free rate and exact guideline concordance, after adjustment for model and scenario complexity. In contrast, performance deteriorated markedly with increasing scenario complexity, with the error burden rising particularly in history-dependent, genotype-sensitive, and multistep decision scenarios. The predominance of undermanagement, genotype misinterpretation, and history neglect as the most frequent error types suggests that the principal vulnerability of LLMs in this domain lies not only in incomplete knowledge but also in insufficient risk-based contextual reasoning.</p>
        <p>These findings indicate that LLMs may have value as structured, clinician-supervised decision support tools in cervical screening management, but that they have not yet reached a sufficient level of safety for autonomous clinical use. Future studies incorporating broader ASCCP coverage, human-AI interactive workflows, and structured verification strategies will be critical for evaluating the safe clinical integration of these tools.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Prespecified gold-standard management decision for each of the 60 clinical scenarios, with its American Society for Colposcopy and Cervical Pathology (ASCCP) source reference and post hoc source verification.</p>
        <media xlink:href="jmir_v28i1e98131_app1.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 24 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Full list of the 60 clinical scenarios used for benchmark evaluation, with category and complexity classification.</p>
        <media xlink:href="jmir_v28i1e98131_app2.docx" xlink:title="DOCX File , 20 KB"/>
      </supplementary-material>
      <supplementary-material id="app3">
        <label>Multimedia Appendix 3</label>
        <p>Verbatim prompt templates used in the 3 study arms.</p>
        <media xlink:href="jmir_v28i1e98131_app3.docx" xlink:title="DOCX File , 16 KB"/>
      </supplementary-material>
      <supplementary-material id="app4">
        <label>Multimedia Appendix 4</label>
        <p>Illustrative scoring example demonstrating the 3-tier classification system.</p>
        <media xlink:href="jmir_v28i1e98131_app4.docx" xlink:title="DOCX File , 17 KB"/>
      </supplementary-material>
      <supplementary-material id="app5">
        <label>Multimedia Appendix 5</label>
        <p>Supplementary tables: interrater reliability, exploratory arm 3, unadjusted response-level CIs, pairwise model comparisons, and sensitivity analyses.</p>
        <media xlink:href="jmir_v28i1e98131_app5.docx" xlink:title="DOCX File , 21 KB"/>
      </supplementary-material>
      <supplementary-material id="app6">
        <label>Multimedia Appendix 6</label>
        <p>Analysis code used to generate the reported statistics, with a README describing which script produces each table.</p>
        <media xlink:href="jmir_v28i1e98131_app6.zip" xlink:title="ZIP File  (Zip Archive), 16 KB"/>
      </supplementary-material>
      <supplementary-material id="app7">
        <label>Multimedia Appendix 7</label>
        <p>CHART checklist.</p>
        <media xlink:href="jmir_v28i1e98131_app7.docx" xlink:title="DOCX File , 42 KB"/>
      </supplementary-material>
      <supplementary-material id="app8">
        <label>Multimedia Appendix 8</label>
        <p>Response classification heatmap by scenario, model, and prompt arm, annotated by complexity tier and coverage group.</p>
        <media xlink:href="jmir_v28i1e98131_app8.png" xlink:title="PNG File , 174 KB"/>
      </supplementary-material>
      <supplementary-material id="app9">
        <label>Multimedia Appendix 9</label>
        <p>Intramodel response consistency across 3 repetitions.</p>
        <media xlink:href="jmir_v28i1e98131_app9.docx" xlink:title="DOCX File , 15 KB"/>
      </supplementary-material>
      <supplementary-material id="app10">
        <label>Multimedia Appendix 10</label>
        <p>Coded observation-level evaluation dataset for the primary analysis and the exploratory arm 3.</p>
        <media xlink:href="jmir_v28i1e98131_app10.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 104 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">AGC</term>
          <def>
            <p>atypical glandular cells</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">ASCCP</term>
          <def>
            <p>American Society for Colposcopy and Cervical Pathology</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">ASC-H</term>
          <def>
            <p>atypical squamous cells, cannot exclude high-grade squamous intraepithelial lesion</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">ASC-US</term>
          <def>
            <p>atypical squamous cells of undetermined significance</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">CHART</term>
          <def>
            <p>Chatbot Assessment Reporting Tool</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">CIN 3+</term>
          <def>
            <p>cervical intraepithelial neoplasia grade 3 or worse</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">ESGO</term>
          <def>
            <p>European Society of Gynaecological Oncology</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">ESP</term>
          <def>
            <p>European Society of Pathology</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">ESTRO</term>
          <def>
            <p>European Society for Radiotherapy and Oncology</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb10">GEE</term>
          <def>
            <p>generalized estimating equation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb11">HPV</term>
          <def>
            <p>human papillomavirus</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb12">HSIL</term>
          <def>
            <p>high-grade squamous intraepithelial lesion</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb13">ICC</term>
          <def>
            <p>intraclass correlation coefficient</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb14">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb15">LSIL</term>
          <def>
            <p>low-grade squamous intraepithelial lesion</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb16">NILM</term>
          <def>
            <p>negative for intraepithelial lesion or malignancy</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb17">OR</term>
          <def>
            <p>odds ratio</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb18">pp</term>
          <def>
            <p>percentage points</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb19">TRIPOD-LLM</term>
          <def>
            <p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis–Large Language Models</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors thank the Ankara Etlik City Hospital Department of Obstetrics and Gynecology for institutional support during the development of this methodological study. This research received no external funding. During the preparation of this manuscript, the authors used generative AI tools for language refinement and formatting assistance only. The authors reviewed and edited the output and take full responsibility for the content of the publication.</p>
    </ack>
    <notes>
      <sec>
        <title>Funding</title>
        <p>The authors declared no financial support was received for this work.</p>
      </sec>
    </notes>
    <notes>
      <sec>
        <title>Data Availability</title>
        <p>The coded observation-level evaluation dataset and the exploratory arm 3 dataset, the gold-standard rubric with its ASCCP source references, the verbatim prompt templates, and the analysis code are provided as <xref ref-type="supplementary-material" rid="app10">Multimedia Appendices 10</xref>, <xref ref-type="supplementary-material" rid="app1">1</xref>, <xref ref-type="supplementary-material" rid="app3">3</xref>, and <xref ref-type="supplementary-material" rid="app6">6</xref>, respectively. The verbatim model response texts are not available. All queries were conducted through publicly available web interfaces in independent, non–account-linked sessions; the response texts were not prospectively archived as part of the study dataset and no recoverable session history was subsequently available. These materials therefore permit reproduction of every reported count, proportion, and test statistic, and inspection of the answer key, but not independent readjudication of the mapping from response text to classification category.</p>
      </sec>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>ÖOE and CE conceptualized the study, developed the methodology, and performed validation. ÖOE conducted the formal analysis, data curation, visualization, supervision, and project administration. ÖOE and CE conducted the investigation. ÖOE prepared the original draft of the manuscript, and ÖOE and CE reviewed and edited the manuscript. All authors have read and agreed to the published version of the manuscript.</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Perkins</surname>
              <given-names>RB</given-names>
            </name>
            <name name-style="western">
              <surname>Guido</surname>
              <given-names>RS</given-names>
            </name>
            <name name-style="western">
              <surname>Castle</surname>
              <given-names>PE</given-names>
            </name>
            <name name-style="western">
              <surname>Chelmow</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Einstein</surname>
              <given-names>MH</given-names>
            </name>
            <name name-style="western">
              <surname>Garcia</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Huh</surname>
              <given-names>WK</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>JJ</given-names>
            </name>
            <name name-style="western">
              <surname>Moscicki</surname>
              <given-names>A-B</given-names>
            </name>
            <name name-style="western">
              <surname>Nayar</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>2019 ASCCP risk-based management consensus guidelines for abnormal cervical cancer screening tests and cancer precursors</article-title>
          <source>J Low Genit Tract Dis</source>
          <year>2020</year>
          <volume>24</volume>
          <issue>2</issue>
          <fpage>102</fpage>
          <lpage>131</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/32243307"/>
          </comment>
          <pub-id pub-id-type="doi">10.1097/LGT.0000000000000525</pub-id>
          <pub-id pub-id-type="medline">32243307</pub-id>
          <pub-id pub-id-type="pii">00128360-202004000-00003</pub-id>
          <pub-id pub-id-type="pmcid">PMC7147428</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Egemen</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Cheung</surname>
              <given-names>LC</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Demarco</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Perkins</surname>
              <given-names>RB</given-names>
            </name>
            <name name-style="western">
              <surname>Kinney</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Poitras</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Befano</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Locke</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Guido</surname>
              <given-names>RS</given-names>
            </name>
            <name name-style="western">
              <surname>Wiser</surname>
              <given-names>AL</given-names>
            </name>
            <name name-style="western">
              <surname>Gage</surname>
              <given-names>JC</given-names>
            </name>
            <name name-style="western">
              <surname>Katki</surname>
              <given-names>HA</given-names>
            </name>
            <name name-style="western">
              <surname>Wentzensen</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Castle</surname>
              <given-names>PE</given-names>
            </name>
            <name name-style="western">
              <surname>Schiffman</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Lorey</surname>
              <given-names>TS</given-names>
            </name>
          </person-group>
          <article-title>Risk estimates supporting the 2019 ASCCP risk-based management consensus guidelines</article-title>
          <source>J Low Genit Tract Dis</source>
          <year>2020</year>
          <volume>24</volume>
          <issue>2</issue>
          <fpage>132</fpage>
          <lpage>143</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/32243308"/>
          </comment>
          <pub-id pub-id-type="doi">10.1097/LGT.0000000000000529</pub-id>
          <pub-id pub-id-type="medline">32243308</pub-id>
          <pub-id pub-id-type="pii">00128360-202004000-00004</pub-id>
          <pub-id pub-id-type="pmcid">PMC7147417</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cheung</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Egemen</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Katki</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Demarco</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wiser</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Perkins</surname>
              <given-names>RB</given-names>
            </name>
            <name name-style="western">
              <surname>Guido</surname>
              <given-names>RS</given-names>
            </name>
            <name name-style="western">
              <surname>Wentzensen</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Schiffman</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>2019 ASCCP risk-based management consensus guidelines: methods for risk estimation, recommended management, and validation</article-title>
          <source>J Low Genit Tract Dis</source>
          <year>2020</year>
          <volume>24</volume>
          <issue>2</issue>
          <fpage>90</fpage>
          <lpage>101</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/32243306"/>
          </comment>
          <pub-id pub-id-type="doi">10.1097/LGT.0000000000000528</pub-id>
          <pub-id pub-id-type="medline">32243306</pub-id>
          <pub-id pub-id-type="pii">00128360-202004000-00002</pub-id>
          <pub-id pub-id-type="pmcid">PMC7147416</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Perkins</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Guido</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Castle</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Chelmow</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Einstein</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Garcia</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Huh</surname>
              <given-names>WK</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>JJ</given-names>
            </name>
            <name name-style="western">
              <surname>Moscicki</surname>
              <given-names>A-B</given-names>
            </name>
            <name name-style="western">
              <surname>Nayar</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Saraiya</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Sawaya</surname>
              <given-names>GF</given-names>
            </name>
            <name name-style="western">
              <surname>Wentzensen</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Schiffman</surname>
              <given-names>M</given-names>
            </name>
            <collab>2019 ASCCP Risk-Based Management Consensus Guidelines Committee</collab>
          </person-group>
          <article-title>2019 ASCCP risk-based management consensus guidelines: updates through 2023</article-title>
          <source>J Low Genit Tract Dis</source>
          <year>2024</year>
          <volume>28</volume>
          <issue>1</issue>
          <fpage>3</fpage>
          <lpage>6</lpage>
          <pub-id pub-id-type="doi">10.1097/LGT.0000000000000788</pub-id>
          <pub-id pub-id-type="medline">38117563</pub-id>
          <pub-id pub-id-type="pii">00128360-202401000-00002</pub-id>
          <pub-id pub-id-type="pmcid">PMC10755815</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wentzensen</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Garcia</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Clarke</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Massad</surname>
              <given-names>LS</given-names>
            </name>
            <name name-style="western">
              <surname>Cheung</surname>
              <given-names>LC</given-names>
            </name>
            <name name-style="western">
              <surname>Egemen</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Guido</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Huh</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Saslow</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Smith</surname>
              <given-names>RA</given-names>
            </name>
            <name name-style="western">
              <surname>Unger</surname>
              <given-names>ER</given-names>
            </name>
            <name name-style="western">
              <surname>Perkins</surname>
              <given-names>RB</given-names>
            </name>
          </person-group>
          <article-title>Enduring consensus guidelines for cervical cancer screening and management: introduction to the scope and process</article-title>
          <source>J Low Genit Tract Dis</source>
          <year>2024</year>
          <volume>28</volume>
          <issue>2</issue>
          <fpage>117</fpage>
          <lpage>123</lpage>
          <pub-id pub-id-type="doi">10.1097/LGT.0000000000000804</pub-id>
          <pub-id pub-id-type="medline">38446573</pub-id>
          <pub-id pub-id-type="pii">00128360-202404000-00001</pub-id>
          <pub-id pub-id-type="pmcid">PMC11520335</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Clarke</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Wentzensen</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Perkins</surname>
              <given-names>RB</given-names>
            </name>
            <name name-style="western">
              <surname>Garcia</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Arrindell</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Chelmow</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Cheung</surname>
              <given-names>LC</given-names>
            </name>
            <name name-style="western">
              <surname>Darragh</surname>
              <given-names>TM</given-names>
            </name>
            <name name-style="western">
              <surname>Egemen</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Guido</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Huh</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>Recommendations for use of p16/Ki67 dual stain for management of individuals testing positive for human papillomavirus</article-title>
          <source>J Low Genit Tract Dis</source>
          <year>2024</year>
          <volume>28</volume>
          <issue>2</issue>
          <fpage>124</fpage>
          <lpage>130</lpage>
          <pub-id pub-id-type="doi">10.1097/LGT.0000000000000802</pub-id>
          <pub-id pub-id-type="medline">38446575</pub-id>
          <pub-id pub-id-type="pii">00128360-202404000-00002</pub-id>
          <pub-id pub-id-type="pmcid">PMC11331430</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Marcus</surname>
              <given-names>JZ</given-names>
            </name>
            <name name-style="western">
              <surname>Cason</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Downs</surname>
              <given-names>LS</given-names>
            </name>
            <name name-style="western">
              <surname>Einstein</surname>
              <given-names>MH</given-names>
            </name>
            <name name-style="western">
              <surname>Flowers</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>The ASCCP cervical cancer screening task force endorsement and opinion on the American Cancer Society updated cervical cancer screening guidelines</article-title>
          <source>J Low Genit Tract Dis</source>
          <year>2021</year>
          <volume>25</volume>
          <issue>3</issue>
          <fpage>187</fpage>
          <lpage>191</lpage>
          <pub-id pub-id-type="doi">10.1097/LGT.0000000000000614</pub-id>
          <pub-id pub-id-type="medline">34138787</pub-id>
          <pub-id pub-id-type="pii">00128360-202107000-00001</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ferreira Santos</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ladeiras-Lopes</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Leite</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Dores</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Applications of large language models in cardiovascular disease: a systematic review</article-title>
          <source>Eur Heart J Digit Health</source>
          <year>2025</year>
          <volume>6</volume>
          <issue>4</issue>
          <fpage>540</fpage>
          <lpage>553</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/ehjdh/article-lookup/doi/10.1093/ehjdh/ztaf028"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/ehjdh/ztaf028</pub-id>
          <pub-id pub-id-type="medline">40703130</pub-id>
          <pub-id pub-id-type="pii">ztaf028</pub-id>
          <pub-id pub-id-type="pmcid">PMC12282349</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gaber</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Shaik</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Allega</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Bilecz</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Busch</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Goon</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Franke</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Akalin</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Evaluating large language model workflows in clinical decision support for triage and referral and diagnosis</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>263</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01684-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01684-1</pub-id>
          <pub-id pub-id-type="medline">40346344</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01684-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC12064692</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Gong</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Gu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Lian</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Mao</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>Zhicheng</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>Lingyun</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>Wenjie</given-names>
            </name>
            <name name-style="western">
              <surname>Ji</surname>
              <given-names>Yajie</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>Yunhui</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Chunbo</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>Yunlu</given-names>
            </name>
            <name name-style="western">
              <surname>Ye</surname>
              <given-names>Qianling</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>Rui</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Mingyu</given-names>
            </name>
            <name name-style="western">
              <surname>Niu</surname>
              <given-names>Lijuan</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Zhihao</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>Peng</given-names>
            </name>
            <name name-style="western">
              <surname>Lang</surname>
              <given-names>Mengran</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Yue</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Huimin</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>Haitao</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Long</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>Qiguang</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Si-Xuan</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Lina</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>Hua</given-names>
            </name>
            <name name-style="western">
              <surname>Ye</surname>
              <given-names>Dongqiang</given-names>
            </name>
            <name name-style="western">
              <surname>Meng</surname>
              <given-names>Lingmin</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>Youtao</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>Naixin</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>Jianxiong</given-names>
            </name>
          </person-group>
          <article-title>A novel evaluation benchmark for medical LLMs illuminating safety and effectiveness in clinical domains</article-title>
          <source>NPJ Digit Med</source>
          <year>2026</year>
          <volume>9</volume>
          <issue>1</issue>
          <fpage>91</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-025-02277-8</pub-id>
          <pub-id pub-id-type="medline">41454006</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-02277-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC12855988</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nacanabo</surname>
              <given-names>MW</given-names>
            </name>
            <name name-style="western">
              <surname>Bayala</surname>
              <given-names>YLT</given-names>
            </name>
            <name name-style="western">
              <surname>Seghda</surname>
              <given-names>AAT</given-names>
            </name>
            <name name-style="western">
              <surname>Tall/Thiam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Yaméogo</surname>
              <given-names>AR</given-names>
            </name>
            <name name-style="western">
              <surname>Yaméogo</surname>
              <given-names>NV</given-names>
            </name>
            <name name-style="western">
              <surname>Samadoulougou</surname>
              <given-names>AK</given-names>
            </name>
            <name name-style="western">
              <surname>Zabsonré</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Comparative study of the performance of ChatGPT-4, Claude, Gemini, Mistral, and Perplexity on multiple-choice questions in cardiology</article-title>
          <source>BMC Cardiovasc Disord</source>
          <year>2026</year>
          <volume>26</volume>
          <issue>1</issue>
          <fpage>32</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmccardiovascdisord.biomedcentral.com/articles/10.1186/s12872-025-05431-y"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12872-025-05431-y</pub-id>
          <pub-id pub-id-type="medline">41366313</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12872-025-05431-y</pub-id>
          <pub-id pub-id-type="pmcid">PMC12802300</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ong</surname>
              <given-names>JCL</given-names>
            </name>
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Elangovan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Lim</surname>
              <given-names>GYS</given-names>
            </name>
            <name name-style="western">
              <surname>Lim</surname>
              <given-names>DYZ</given-names>
            </name>
            <name name-style="western">
              <surname>Sng</surname>
              <given-names>GGR</given-names>
            </name>
            <name name-style="western">
              <surname>Ke</surname>
              <given-names>YH</given-names>
            </name>
            <name name-style="western">
              <surname>Tung</surname>
              <given-names>JYM</given-names>
            </name>
            <name name-style="western">
              <surname>Zhong</surname>
              <given-names>RJ</given-names>
            </name>
            <name name-style="western">
              <surname>Koh</surname>
              <given-names>CMY</given-names>
            </name>
          </person-group>
          <article-title>Large language model as clinical decision support system augments medication safety in 16 clinical specialties</article-title>
          <source>Cell Rep Med</source>
          <year>2025</year>
          <volume>6</volume>
          <issue>10</issue>
          <fpage>102323</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2666-3791(25)00396-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.xcrm.2025.102323</pub-id>
          <pub-id pub-id-type="medline">40997804</pub-id>
          <pub-id pub-id-type="pii">S2666-3791(25)00396-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC12629785</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Stelling</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Brink</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Grieb</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Kraus</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Güler</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>Reliability and performance stability of large language models in medical knowledge assessment: evidence from the European Board of Nuclear Medicine Examination</article-title>
          <source>AI (Basel)</source>
          <year>2026</year>
          <volume>7</volume>
          <issue>2</issue>
          <fpage>77</fpage>
          <pub-id pub-id-type="doi">10.3390/ai7020077</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Güler</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Grieb</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Kraus</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Lautenbach</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Stelling</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Diagnostic accuracy and stability of multimodal large language models for hand fracture detection: a multi-run evaluation on plain radiographs</article-title>
          <source>Diagnostics (Basel)</source>
          <year>2026</year>
          <volume>16</volume>
          <issue>3</issue>
          <fpage>424</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=diagnostics16030424"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/diagnostics16030424</pub-id>
          <pub-id pub-id-type="medline">41681742</pub-id>
          <pub-id pub-id-type="pii">diagnostics16030424</pub-id>
          <pub-id pub-id-type="pmcid">PMC12897326</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yurtcu</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Ozvural</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Keyif</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Analyzing the performance of ChatGPT in answering inquiries about cervical cancer</article-title>
          <source>Int J Gynaecol Obstet</source>
          <year>2025</year>
          <volume>168</volume>
          <issue>2</issue>
          <fpage>502</fpage>
          <lpage>507</lpage>
          <pub-id pub-id-type="doi">10.1002/ijgo.15861</pub-id>
          <pub-id pub-id-type="medline">39148482</pub-id>
          <pub-id pub-id-type="pmcid">PMC11726164</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kuerbanjiang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Jiamaliding</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yi</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Performance evaluation of large language models in cervical cancer management based on a standardized questionnaire: comparative study</article-title>
          <source>J Med Internet Res</source>
          <year>2025</year>
          <volume>27</volume>
          <fpage>e63626</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2025//e63626/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/63626</pub-id>
          <pub-id pub-id-type="medline">39908540</pub-id>
          <pub-id pub-id-type="pii">v27i1e63626</pub-id>
          <pub-id pub-id-type="pmcid">PMC11840365</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Angyal</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Bertalan</surname>
              <given-names>Á</given-names>
            </name>
            <name name-style="western">
              <surname>Domján</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Dinya</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Exploring the possibilities and limitations of customized large language model to support and improve cervical cancer screening</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2025</year>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>242</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-025-03088-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-025-03088-3</pub-id>
          <pub-id pub-id-type="medline">40597085</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-025-03088-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC12220158</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Pavone</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Innocenzi</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Macellari</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Cantarini</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Criscione</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Rosati</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Lecointre</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Carcagnì</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Costantini</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Marescaux</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Assessing the accuracy of large language models on European guidelines for cervical cancer: an in silico benchmarking study</article-title>
          <source>BJOG</source>
          <year>2026</year>
          <volume>133</volume>
          <issue>4</issue>
          <fpage>771</fpage>
          <lpage>778</lpage>
          <pub-id pub-id-type="doi">10.1111/1471-0528.70095</pub-id>
          <pub-id pub-id-type="medline">41287196</pub-id>
          <pub-id pub-id-type="pmcid">PMC12884235</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>KY</given-names>
            </name>
            <name name-style="western">
              <surname>Zeger</surname>
              <given-names>SL</given-names>
            </name>
          </person-group>
          <article-title>Longitudinal data analysis using generalized linear models</article-title>
          <source>Biometrika</source>
          <year>1986</year>
          <volume>73</volume>
          <issue>1</issue>
          <fpage>13</fpage>
          <lpage>22</lpage>
          <pub-id pub-id-type="doi">10.1093/biomet/73.1.13</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Landis</surname>
              <given-names>JR</given-names>
            </name>
            <name name-style="western">
              <surname>Koch</surname>
              <given-names>GG</given-names>
            </name>
          </person-group>
          <article-title>The measurement of observer agreement for categorical data</article-title>
          <source>Biometrics</source>
          <year>1977</year>
          <volume>33</volume>
          <issue>1</issue>
          <fpage>159</fpage>
          <lpage>174</lpage>
          <pub-id pub-id-type="medline">843571</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Firth</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Bias reduction of maximum likelihood estimates</article-title>
          <source>Biometrika</source>
          <year>1993</year>
          <volume>80</volume>
          <issue>1</issue>
          <fpage>27</fpage>
          <lpage>38</lpage>
          <pub-id pub-id-type="doi">10.1093/biomet/80.1.27</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Heinze</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Schemper</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>A solution to the problem of separation in logistic regression</article-title>
          <source>Stat Med</source>
          <year>2002</year>
          <volume>21</volume>
          <issue>16</issue>
          <fpage>2409</fpage>
          <lpage>2419</lpage>
          <pub-id pub-id-type="doi">10.1002/sim.1047</pub-id>
          <pub-id pub-id-type="medline">12210625</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hoenig</surname>
              <given-names>JM</given-names>
            </name>
            <name name-style="western">
              <surname>Heisey</surname>
              <given-names>DM</given-names>
            </name>
          </person-group>
          <article-title>The abuse of power</article-title>
          <source>Am Stat</source>
          <year>2001</year>
          <volume>55</volume>
          <issue>1</issue>
          <fpage>19</fpage>
          <lpage>24</lpage>
          <pub-id pub-id-type="doi">10.1198/000313001300339897</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <collab>CHART Collaborative</collab>
            <name name-style="western">
              <surname>Huo</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Chartash</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Thirunavukarasu</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Flanagin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Iorio</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cacciamani</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>Reporting guideline for chatbot health advice studies: the CHART statement</article-title>
          <source>Artif Intell Med</source>
          <year>2025</year>
          <volume>168</volume>
          <fpage>103222</fpage>
          <pub-id pub-id-type="doi">10.1016/j.artmed.2025.103222</pub-id>
          <pub-id pub-id-type="medline">40753040</pub-id>
          <pub-id pub-id-type="pii">S0933-3657(25)00157-5</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gallifant</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Afshar</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ameen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Aphinyanaphongs</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cacciamani</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Dligach</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Daneshjou</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Fernandes</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>The TRIPOD-LLM reporting guideline for studies using large language models</article-title>
          <source>Nat Med</source>
          <year>2025</year>
          <volume>31</volume>
          <issue>1</issue>
          <fpage>60</fpage>
          <lpage>69</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-024-03425-5</pub-id>
          <pub-id pub-id-type="medline">39779929</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-024-03425-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC12104976</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Saraiya</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Colbert</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Bhat</surname>
              <given-names>GL</given-names>
            </name>
            <name name-style="western">
              <surname>Almonte</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Winters</surname>
              <given-names>DW</given-names>
            </name>
            <name name-style="western">
              <surname>Sebastian</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>O'Hanlon</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Meadows</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Nosal</surname>
              <given-names>MR</given-names>
            </name>
            <name name-style="western">
              <surname>Richards</surname>
              <given-names>TB</given-names>
            </name>
          </person-group>
          <article-title>Computable guidelines and clinical decision support for cervical cancer screening and management to improve outcomes and health equity</article-title>
          <source>J Womens Health (Larchmt)</source>
          <year>2022</year>
          <volume>31</volume>
          <issue>4</issue>
          <fpage>462</fpage>
          <lpage>468</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/35467443"/>
          </comment>
          <pub-id pub-id-type="doi">10.1089/jwh.2022.0100</pub-id>
          <pub-id pub-id-type="medline">35467443</pub-id>
          <pub-id pub-id-type="pmcid">PMC9206487</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
