<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e84915</article-id><article-id pub-id-type="doi">10.2196/84915</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Evaluating the Accuracy of Large Language Models in Risk-of-Bias Assessment Using Version 2 of the Cochrane Risk-of-Bias Tool for Randomized Trials: Exploratory Feasibility Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Lai</surname><given-names>Yu-Ju</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lin</surname><given-names>Shen-Hua</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Liu</surname><given-names>Jen-Wei</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Pharmacy, Fu Jen Catholic University Hospital, Fu Jen Catholic University</institution><addr-line>No.69, Guizi Rd., Taishan Dist.</addr-line><addr-line>New Taipei City</addr-line><country>Taiwan</country></aff><aff id="aff2"><institution>Graduate Institute of Clinical Medicine, College of Medicine, Taipei Medical University</institution><addr-line>Taipei</addr-line><country>Taiwan</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Ting</surname><given-names>Eon</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Franchi</surname><given-names>Eriberto</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Jen-Wei Liu, PhD, Department of Pharmacy, Fu Jen Catholic University Hospital, Fu Jen Catholic University, No.69, Guizi Rd., Taishan Dist., New Taipei City, 24352, Taiwan; <email>jerryljw@gmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>25</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e84915</elocation-id><history><date date-type="received"><day>29</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>11</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>12</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Yu-Ju Lai, Shen-Hua Lin, Jen-Wei Liu. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 25.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e84915"/><abstract><sec><title>Background</title><p>Large language models (LLMs) have the potential to improve the efficiency of evidence synthesis, but their reliability in performing complex tasks such as risk-of-bias (ROB) assessment in randomized controlled trials (RCTs) remains unclear.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate whether LLMs can reliably assess ROB in RCTs using version 2 of the Cochrane ROB tool for randomized trials (ROB 2).</p></sec><sec sec-type="methods"><title>Methods</title><p>This study was conducted between December 28, 2024, and February 28, 2025, in adherence to American Association for Public Opinion Research reporting guidelines. Twenty-nine RCTs were selected from published Cochrane systematic reviews across diverse medical fields. We developed a structured prompt engineering framework that transformed ROB 2 decision trees into logical rules for the LLM. Each RCT was independently evaluated twice by ChatGPT, with Cochrane review authors&#x2019; assessments serving as the reference standard for comparison. The main outcomes were the accuracy and consistency of ROB 2 assessments at both the domain and trial levels, evaluated using accuracy, sensitivity, specificity, and <italic>F</italic><sub>1</sub>-score. Consistency between the repeated assessments was quantified using the Cohen &#x03BA; and prevalence-adjusted, bias-adjusted &#x03BA;.</p></sec><sec sec-type="results"><title>Results</title><p>The LLM demonstrated a moderate aggregate domain accuracy of 73.1% (95% CI 64.7%-81.5%) in the first assessment and 75.9% (95% CI 66.3%-85.4%) in the second assessment. Domain-averaged sensitivity decreased from 61.4% (95% CI 48.1%&#x2010;74.7%) to 53.4% (95% CI 41.7%&#x2010;65.0%), whereas domain-averaged specificity increased from 75.8% (95% CI 65.3%-86.3%) to 81.1% (95%CI 67.1%-95%), indicating a conservative tendency in identifying a high ROB. Domain-level accuracy ranged from 62.1% to 87.9%, with the lowest accuracy observed in domain 1 and the lowest <italic>F</italic><sub>1</sub>-scores observed in domain 2. Consistency between repeated assessments was high, with a mean agreement of 89.0% (SD 7.5%), and Cohen &#x03BA; values were 0.86, 0.39, 0.56, 0.84, and 0.85 in domains 1 to 5, respectively.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In this exploratory study, ChatGPT demonstrated moderate accuracy and high consistency in assessing ROB in RCTs using the ROB 2. However, its reliability diminished in complex scenarios requiring interpretation of implicit narratives or behavioral nuance. These findings suggest that LLMs may support methodological evaluations in systematic reviews by acting as automated screeners to reduce reviewer burden, but current implementation still requires expert oversight, particularly for trials involving subjective outcomes or nonstandard reporting.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>LLMs</kwd><kwd>risk of bias</kwd><kwd>version 2 of the Cochrane risk-of-bias tool for randomized trials</kwd><kwd>ROB 2</kwd><kwd>ChatGPT</kwd><kwd>randomized controlled trial</kwd><kwd>RCT</kwd><kwd>systematic reviews</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Systematic reviews play a foundational role in evidence-based medicine by synthesizing findings from primary studies to inform clinical decision-making, guideline development, and health policy [<xref ref-type="bibr" rid="ref1">1</xref>]. As the volume of biomedical literature continues to grow rapidly, the demand for efficient, high-quality evidence synthesis has become more urgent. Among the key components of systematic reviews is the assessment of the risk of bias (ROB), particularly in randomized controlled trials (RCTs), which serve as the primary source of evidence for many clinical guidelines. The Grading of Recommendations Assessment, Development, and Evaluation (GRADE) framework emphasizes that the certainty of evidence is closely tied to the ROB in included studies within a systematic review [<xref ref-type="bibr" rid="ref2">2</xref>].</p><p>To support rigorous and transparent ROB assessments, the Cochrane Collaboration developed version 2 of its ROB tool for randomized trials (ROB 2), which evaluates bias at the outcome level across 5 domains: randomization process, deviations from intended interventions, missing outcome data, measurement of the outcome, and selection of the reported results [<xref ref-type="bibr" rid="ref3">3</xref>]. The tool relies on signaling questions and structured algorithms to facilitate reproducible judgments. Despite its methodological strengths, applying the ROB 2 remains time and resource intensive, requiring trained reviewers and substantial manual effort.</p><p>Large language models (LLMs), with their advanced capabilities in natural language understanding and reasoning, have been proposed as tools to automate complex tasks in medical text analysis [<xref ref-type="bibr" rid="ref4">4</xref>]. Preliminary research suggests that LLMs may be able to interpret trial reports and mimic human judgment in various evaluative tasks. However, whether LLMs can perform structured ROB assessments in line with established tools such as the ROB 2 remains an open question [<xref ref-type="bibr" rid="ref5">5</xref>]. To address this gap, we conducted a study to evaluate the feasibility, accuracy, and consistency of using LLMs to perform ROB assessments for RCTs guided by a structured prompt based on the ROB 2 framework.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>This study was conducted between December 28, 2024, and February 28, 2025, in adherence to the American Association for Public Opinion Research reporting guidelines [<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>For this study, we used the GPT-o3-mini-high model via the web interface. OpenAI&#x2019;s GPT-o3-mini has been optimized for science, technology, engineering, and mathematics reasoning, with medium reasoning effort matching the performance of GPT-o1 in math, coding, and science while delivering faster responses [<xref ref-type="bibr" rid="ref7">7</xref>]. Due to the technical constraints of the ChatGPT web interface, all assessments were performed using the model&#x2019;s default settings as manual adjustments to parameters such as temperature and top_p are not supported in this environment. This model was used to systematically assess the ROB based on the ROB 2 in RCTs using its reasoning capabilities to simulate the evaluation process conducted in systematic reviews.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>We used only publicly available published literature and LLMs without involving any personal or identifiable private data. This study followed the &#x201C;Scope of Human Research Projects Exempt from Institutional Review Board Review&#x201D; (1010265075) [<xref ref-type="bibr" rid="ref8">8</xref>] issued by the Ministry of Health and Welfare, Taiwan. As this study relied solely on legally and publicly disclosed information, it was exempt from ethical review or informed consent requirements.</p></sec><sec id="s2-3"><title>Model Training and Prompt Development</title><p>Prompt development followed the logic and structure outlined in the ROB 2 guidance document [<xref ref-type="bibr" rid="ref9">9</xref>]. Systematic prompt engineering techniques were developed to create structured prompts, allowing ChatGPT to perform tasks.</p><p>We used 3 RCTs as pilot data, which were excluded from the final analytic sample. An iterative refinement process was used referencing published systematic reviews as a reference standard. On the basis of the framework by Lai et al [<xref ref-type="bibr" rid="ref5">5</xref>], we defined the LLMs&#x2019; character, provided foundational definitions of the ROB 2 domains, and systematically transformed the official ROB 2 decision trees into logical rules within the prompt. When model-generated ROB judgments differed from the original assessments, prompts were revised and tested until alignment with expert assessments was consistently achieved. <xref ref-type="fig" rid="figure1">Figure 1</xref> shows the main study process, and the final prompt can be found in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flow diagram of the main study process. LLM: large language model; RCT: randomized controlled trial; ROB 2: version 2 of the Cochrane risk-of-bias tool for randomized trials; SR: systematic review.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84915_fig01.png"/></fig></sec><sec id="s2-4"><title>Selection of Sample</title><p>We conducted a search of the PubMed database to identify Cochrane systematic reviews that used the ROB 2 for assessing ROB in RCTs. Two reviewers (YJL and JWL) independently screened the full texts of the retrieved systematic reviews to determine eligibility. Reviews were excluded if they did not provide detailed ROB assessments, such as an ROB table. From the eligible 178 Cochrane systematic reviews, we first randomized the identified reviews using numbers generated via Google Sheets and sequentially screened them until a pooled baseline of over 100 RCTs was established. Second, using the same random number generation method, we selected 30 RCTs from this pool for analysis to mitigate potential clustering effects. These 30 RCTs were distributed across 11 reviews. Of the 30 RCTs, 1 (3.3%) was excluded due to incomplete reporting of ROB assessment, resulting in a final analytic sample of 29 (96.7%) RCTs. Notably, one of the included trials, that by Maher et al [<xref ref-type="bibr" rid="ref10">10</xref>], was retracted after our study period [<xref ref-type="bibr" rid="ref11">11</xref>]. Because its Cochrane assessment contributed to our reference standard, we retained this trial in the primary analytic sample but conducted a sensitivity check to evaluate its impact.</p></sec><sec id="s2-5"><title>Application of ChatGPT for ROB Assessment</title><p>We used the GPT-o3-mini-high model for its strong reasoning capabilities, enabling consistent and accurate application of structured prompts in the ROB assessment process.</p><p>Each RCT was evaluated using a new ChatGPT conversation to ensure a clean contextual slate. The reviewers defined the primary outcome, provided domain-specific criteria, and uploaded the full trial text. ChatGPT then assessed ROB across the 5 domains in the ROB 2 framework, offering domain-level judgments with rationale (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Each trial was assessed twice under identical conditions. In cases of system interruption, the session was discarded and repeated. A standardized interaction protocol was followed to maximize consistency and reproducibility.</p></sec><sec id="s2-6"><title>Establishment of the Reference Standard</title><p>The criterion standard for comparison was the ROB assessment reported by the Cochrane systematic review authors using the ROB 2 (Table S1 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). These Cochrane assessments were selected for their rigorous methodology and multidisciplinary collaboration, which minimize bias and ensure clinical diversity [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. When apparent errors were identified (eg, typographical mistakes), authors were contacted for clarification. If no response was received, the original Cochrane systematic review assessments were retained as the reference standard to ensure objectivity. A sensitivity analysis was then performed to evaluate the impact of potential errors using data that were manually corrected based on research team consensus.</p></sec><sec id="s2-7"><title>Discrepancy Categorization and Analysis</title><p>To evaluate the discrepancies between the LLM and the reference standard, we conducted an error analysis. Two researchers (YJL and JWL) independently categorized the discrepancies by comparing the LLM-generated rationales against the Cochrane reviewers&#x2019; evidence. Any disagreements were resolved through consensus.</p><p>Discrepancies were categorized as data extraction differences if there was a fundamental mismatch in the explicit information or evidence retrieved from the RCT between the LLM and the Cochrane reviewers. Conversely, they were classified as judgment differences if the LLM accurately extracted the same key points as the human reviewers but applied a different logical interpretation leading to a conflicting judgment.</p></sec><sec id="s2-8"><title>Statistical Analysis</title><p>The LLM was prompted to answer individual signaling questions using &#x201C;yes,&#x201D; &#x201C;probably yes,&#x201D; &#x201C;no,&#x201D; &#x201C;probably no,&#x201D; and &#x201C;no information.&#x201D; Signaling responses were then operationally mapped to the official &#x201C;low risk&#x201D; and &#x201C;high risk&#x201D; categories following the ROB 2 guideline. ROB domain judgments were categorized as follows: &#x201C;low risk&#x201D; was defined as a negative outcome, whereas &#x201C;some concerns&#x201D; and &#x201C;high risk&#x201D; were grouped as a positive outcome. On the basis of this classification, we calculated true positives (TPs), true negatives (TNs), false positives (FPs), and false negatives (FNs).</p><p>Model performance was evaluated using accuracy, sensitivity, specificity, precision, and <italic>F</italic><sub>1</sub>-score. For domain-level analyses, sensitivity and specificity were macroaveraged across domains. Metrics were defined as follows:</p><p>Accuracy = (TPs + TNs)/total number of assessments</p><p>Sensitivity = TPs/(TPs + FNs)</p><p>Specificity = TNs/(TNs + FPs)</p><p>Precision = TPs/(TPs + FPs)</p><p><italic>F</italic><sub>1</sub>-score = 2 &#x00D7; (precision &#x00D7; sensitivity)/(precision + sensitivity)</p><p>To assess internal consistency, we computed the Cohen &#x03BA; and the prevalence-adjusted, bias-adjusted (PABA) &#x03BA;:</p><p>Cohen &#x03BA;&#x2009;=&#x2009;(<italic>P</italic><sub><italic>o</italic></sub>&#x2009;&#x2212;&#x2009;<italic>P</italic><sub><italic>e</italic></sub>)/(1&#x2009;&#x2212;&#x2009;<italic>P</italic><sub><italic>e</italic></sub>)</p><p>PABA &#x03BA; = 2 &#x00D7; <italic>P</italic><sub><italic>o</italic></sub> &#x2013; 1</p><p>In these equations, <italic>P<sub>o</sub></italic>&#x2009;is calculated as&#x2009;(number of agreements on positive&#x2009;+&#x2009;number of agreements on negative)/total number of assessments.</p><p><italic>P</italic><sub><italic>e</italic></sub> = [(<italic>P</italic><sub>1</sub> &#x00D7; <italic>P</italic><sub>2</sub>) + (<italic>N</italic><sub>1</sub> &#x00D7; <italic>N</italic><sub>2</sub>)]/(total number of assessments)<sup>2</sup></p><p>For trials in which the model provided identical ratings across both assessments, the Cohen &#x03BA; was mathematically undefined due to zero variance, which resulted in a zero denominator in the &#x03BA; calculation and was therefore reported as &#x201C;Not available.&#x201D; The agreement thresholds in Table S2 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> followed standard interpretations: &#x03BA; values of 0.81 to 0.99 indicated near-perfect agreement. Additionally, exploratory subgroup analyses by clinical discipline were performed to investigate potential variability in model performance across different medical contexts. Trials were categorized based on the primary medical focus of the RCT and the scope of the parent Cochrane review from which the trial was extracted. In cases in which a trial potentially spanned multiple disciplines, the final categorization was determined through discussion and consensus between 2 authors (YJL and JWL). All statistical analyses were conducted using Google Sheets.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Characteristics of the RCTs</title><p>The final dataset included 29 RCTs [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref41">41</xref>] extracted from 11 Cochrane systematic reviews [<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref52">52</xref>]. These trials represented a spectrum of medical disciplines, including cardiology (n=5, 17.2%), psychology (n=6, 20.7%), infectious diseases (n=6, 20.7%), gastroenterology (n=5, 17.2%), pulmonology (n=4, 13.8%), bone health (n=1, 3.4%), and obstetrics and gynecology (n=2, 6.9%). Primary efficacy-related outcomes varied across studies, including binary outcomes (n=18, 62.1%), with most focusing on all-cause mortality and asthma exacerbation or treatment success. Continuous outcomes (n=11, 37.9%) mainly included pain, physical function, and quality of life. The trials were published between 2000 and 2024. All RCTs were published in English and assessed for ROB using the ROB 2, ensuring a standardized approach to bias evaluation.</p></sec><sec id="s3-2"><title>Accuracy</title><sec id="s3-2-1"><title>Aggregate Domain and Overall Accuracy</title><p>The LLM underwent 2 independent assessments for ROB evaluation, as shown in Table S3 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. The aggregate domain accuracy (<xref ref-type="table" rid="table1">Table 1</xref> and <xref ref-type="fig" rid="figure2">Figure 2</xref>) was comparable between assessments at 73.1% (95% CI 64.7%-81.5%) for the first and 75.9% (95% CI 66.3%-85.4%) for the second, reflecting a marginal improvement in accuracy in the second assessment (relative difference [RD] 2.8%, 95% CI &#x2013;3.1% to 8.6%).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Domain-specific accuracy of assessments.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">TPs<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>, n</td><td align="left" valign="bottom">TNs<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>, n</td><td align="left" valign="bottom">FPs<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>, n</td><td align="left" valign="bottom">FNs<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup>, n</td><td align="left" valign="bottom">Accuracy, %</td><td align="left" valign="bottom">Sensitivity, %</td><td align="left" valign="bottom">Specificity, %</td><td align="left" valign="bottom">Precision, %</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score, %</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="10">Domain 1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assessment 1</td><td align="left" valign="top">5</td><td align="left" valign="top">13</td><td align="left" valign="top">9</td><td align="left" valign="top">2</td><td align="left" valign="top">62.1</td><td align="left" valign="top">71.4</td><td align="left" valign="top">59.1</td><td align="left" valign="top">35.7</td><td align="left" valign="top">47.6</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assessment 2</td><td align="left" valign="top">5</td><td align="left" valign="top">13</td><td align="left" valign="top">9</td><td align="left" valign="top">2</td><td align="left" valign="top">62.1</td><td align="left" valign="top">71.4</td><td align="left" valign="top">59.1</td><td align="left" valign="top">35.7</td><td align="left" valign="top">47.6</td></tr><tr><td align="left" valign="top" colspan="10">Domain 2</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assessment 1</td><td align="left" valign="top">3</td><td align="left" valign="top">17</td><td align="left" valign="top">5</td><td align="left" valign="top">4</td><td align="left" valign="top">69.0</td><td align="left" valign="top">42.9</td><td align="left" valign="top">77.3</td><td align="left" valign="top">37.5</td><td align="left" valign="top">40.0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assessment 2</td><td align="left" valign="top">3</td><td align="left" valign="top">21</td><td align="left" valign="top">1</td><td align="left" valign="top">4</td><td align="left" valign="top">82.8</td><td align="left" valign="top">42.9</td><td align="left" valign="top">95.5</td><td align="left" valign="top">75.0</td><td align="left" valign="top">54.6</td></tr><tr><td align="left" valign="top" colspan="10">Domain 3</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assessment 1</td><td align="left" valign="top">4</td><td align="left" valign="top">19</td><td align="left" valign="top">5</td><td align="left" valign="top">1</td><td align="left" valign="top">79.3</td><td align="left" valign="top">80.0</td><td align="left" valign="top">79.2</td><td align="left" valign="top">44.4</td><td align="left" valign="top">57.1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assessment 2</td><td align="left" valign="top">2</td><td align="left" valign="top">20</td><td align="left" valign="top">4</td><td align="left" valign="top">3</td><td align="left" valign="top">75.9</td><td align="left" valign="top">40.0</td><td align="left" valign="top">83.3</td><td align="left" valign="top">33.3</td><td align="left" valign="top">36.4</td></tr><tr><td align="left" valign="top" colspan="10">Domain 4</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assessment 1</td><td align="left" valign="top">2</td><td align="left" valign="top">23</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">86.2</td><td align="left" valign="top">50.0</td><td align="left" valign="top">92.0</td><td align="left" valign="top">50.0</td><td align="left" valign="top">50.0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assessment 2</td><td align="left" valign="top">2</td><td align="left" valign="top">24</td><td align="left" valign="top">1</td><td align="left" valign="top">2</td><td align="left" valign="top">89.7</td><td align="left" valign="top">50.0</td><td align="left" valign="top">96.0</td><td align="left" valign="top">66.7</td><td align="left" valign="top">57.1</td></tr><tr><td align="left" valign="top" colspan="10">Domain 5</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assessment 1</td><td align="left" valign="top">5</td><td align="left" valign="top">15</td><td align="left" valign="top">6</td><td align="left" valign="top">3</td><td align="left" valign="top">69.0</td><td align="left" valign="top">62.5</td><td align="left" valign="top">71.4</td><td align="left" valign="top">45.5</td><td align="left" valign="top">52.6</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Assessment 2</td><td align="left" valign="top">5</td><td align="left" valign="top">15</td><td align="left" valign="top">6</td><td align="left" valign="top">3</td><td align="left" valign="top">69.0</td><td align="left" valign="top">62.5</td><td align="left" valign="top">71.4</td><td align="left" valign="top">45.5</td><td align="left" valign="top">52.6</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>TP: true positive.</p></fn><fn id="table1fn2"><p><sup>b</sup>TN: true negative.</p></fn><fn id="table1fn3"><p><sup>c</sup>FP: false positive.</p></fn><fn id="table1fn4"><p><sup>d</sup>FN: false negative.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Radar chart of accuracy in domains.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84915_fig02.png"/></fig><p>In contrast, the overall ROB accuracy, derived from the final overall ROB judgment for each trial, was 79.3% (23/29) in the first assessment and 86.2% (25/29) in the second.</p><p>Sensitivity, reflecting the ability to detect high-risk judgments, was higher for the first assessment at 61.4% (95% CI 48.1%-74.7%) compared to 53.4% (95% CI 41.7%-65.0%) in the second, suggesting slightly reduced effectiveness in TP identification by the second assessment (RD 8.0%, 95% CI &#x2013;7.7% to 23.7%). Specificity remained high in both assessments, increasing from 75.8% (95% CI 65.3%-86.3%) in the first to 81.1% (95% CI 67.1%-95.0%) in the second, demonstrating strong performance in identifying low-risk judgments. <italic>F</italic><sub>1</sub>-scores were similar between assessments, with the first one slightly lower at 49.5% (95% CI 43.9%-55.1%) compared to 49.7% (95% CI 42.5%-56.9%) in the second.</p></sec><sec id="s3-2-2"><title>Domain-Specific Accuracy</title><p>Across the 5 ROB 2 domains, the mean accuracy was 74.5% (95% CI 66.0%-83.0%) based on pooled results from 2 assessments. The lowest accuracy was observed in domain 1 (randomization process) at 62.1%, whereas the highest was observed in domain 4 (outcome measurement) at 87.9%. A total of 74 discrepancies were identified in ROB assessments, with 55 (74.3%) resulting from differences in judgments between the LLM and the reference standard and 19 (25.7%) arising from inconsistencies in data extraction. Domain 1 (randomization process) exhibited the highest number of total discrepancies (22/74, 29.7%), with 68.2% (15/22) related to judgment and 31.8% (7/22) related to data extraction. Domain 5 (selection of the reported results) accounted for 24.3% (18/74) of discrepancies, including 66.7% (12/18) judgment-related and 33.3% (6/18) extraction-related differences. Domains 2 and 3 showed intermediate discrepancy rates (14/74, 18.9% and 13/74, 17.6%, respectively), whereas domain 4 exhibited the fewest discrepancies (7/74, 9.5%), predominantly due to judgment differences.</p><p>Sensitivity ranged from 42.9% in domain 2 to 71.4% in domain 1, highlighting variability in identifying high-risk judgments. Specificity was consistently high (range 59.1%-94%), indicating reliable identification of low-risk judgments. The <italic>F</italic><sub>1</sub>-score was highest in domain 4 (53.3%) and lowest in domain 2 (46.2%), highlighting the influence of both sensitivity and precision on performance in nuanced domains.</p></sec><sec id="s3-2-3"><title>Trial-Specific Accuracy</title><p>Across 58 assessments for the 29 trials (Table S4 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>), 7 (24.1%) trials achieved full accuracy (100% correct), whereas 7 (24.1%) trials attained an average accuracy between 80% and 90%. In total, 51.7% (15/29) of the trials had an accuracy of 70% or lower, with the lowest at 40% (<xref ref-type="fig" rid="figure3">Figure 3</xref> [<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref52">52</xref>]).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Heat map of accuracy across trial- and domain-specific contexts [<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref52">52</xref>]. RCT: randomized controlled trial; SR: systematic review.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84915_fig03.png"/></fig><p>Performance varied across medical disciplines, although these findings were limited by small subgroup sample sizes. The reported counts represent aggregated domain-level assessments across 5 ROB 2 domains and 2 assessment cycles per trial. For bone health (1/29, 3.4%), the single trial was correctly judged in both assessments. For other disciplines, we observed the following aggregated domain-level accuracies: in gastroenterology, 39 out of 50 domains were correctly assessed (across 5/29, 17.2% of the trials); in infectious diseases, 48 out of 60 domains were correctly assessed (across 6/29, 20.7% of the trials); in pulmonology, 35 out of 40 domains were correctly assessed (across 4/29, 13.8% of the trials); and in obstetrics and gynecology, 17 out of 20 domains were correctly assessed (across 2/29, 6.9% of the trials). Cardiology and psychology exhibited the lowest accuracy, with 31 out of 50 domains (across 5/29, 17.2% of the trials) and 36 out of 60 domains (across 6/29, 20.7% of the trials) correctly assessed, respectively, with no domains in either discipline achieving full accuracy in the first or second assessment.</p></sec></sec><sec id="s3-3"><title>Sensitivity Analysis</title><p>Notably, a typographical error in the initial assessments by Toouli et al [<xref ref-type="bibr" rid="ref20">20</xref>] was identified in domains 2, 3, and 4. After attempting to contact the study authors without receiving a response, the research team conducted a sensitivity analysis to evaluate the impact of manual adjudications on the reference standard. Correction of the typographical error led to improvements in domain-specific performance. In the first assessment, <italic>F</italic><sub>1</sub>-scores increased from 40.0% to 50.0% for domain 2 and from 57.1% to 61.5% for domain 3, whereas accuracy increased from 69.0% to 72.4% for domain 2 and from 79.3% to 82.8% for domain 3. In the second assessment, while the <italic>F</italic><sub>1</sub>-score for domain 3 improved from 36.4% to 40.0% and its accuracy rose from 75.9% to 79.3%, the <italic>F</italic><sub>1</sub>-score for domain 2 paradoxically decreased from 54.6% to 50.0%, and accuracy decreased from 82.8% to 79.3%. This decrease occurred because the LLM&#x2019;s original judgment in assessment 2 happened to align with the uncorrected Cochrane reference. After manual correction of the reference standard to reflect the true methodological assessment, the LLM&#x2019;s response was reclassified as discrepant. Additionally, we performed a sensitivity check by excluding the subsequently retracted trial by Maher et al [<xref ref-type="bibr" rid="ref10">10</xref>]. Removing this single trial resulted in a slight decrease in the aggregate domain accuracy, shifting from 73.1% to 72.9% in the first assessment and from 75.9% to 75% in the second. These changes were of less than 1 percentage point and did not materially alter the overall trends or conclusions regarding the LLM&#x2019;s performance. Following exclusion of this trial, obstetrics and gynecology was represented by 1 remaining trial in which 8 of 10 domains across the 2 assessments were correctly classified; this result was therefore reported descriptively.</p></sec><sec id="s3-4"><title>Consistency</title><p>The assessment rate (observed agreement; <italic>P</italic><sub><italic>o</italic></sub>) demonstrated high consistency across domains, with a mean of 89.0% (SD 7.5%), ranging from 79.3% to 96.6%. Cohen &#x03BA; values indicated near-perfect agreement in domains 1, 4, and 5 (&#x03BA;=0.86, 0.84, and 0.85, respectively); moderate agreement in domain 3 (&#x03BA;=0.56); and fair agreement in domain 2 (&#x03BA;=0.39). PABA &#x03BA; values followed similar trends (Table S5 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p><p>At the trial level, of all 29 RCT assessments, 17 (58.6%) achieved full consistency (proportion of agreement; <italic>P<sub>o</sub></italic>=1.00) across all domains, whereas 9 (31%) attained a proportion of agreement of exactly 0.80.</p><p>The Cohen &#x03BA; was reported as &#x201C;not available&#x201D; due to zero variance in 24.1% (7/29) of the trials. The mean agreement across all studies was 0.89 (SD 0.16), reflecting high consistency (Table S6 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this study, we used GPT-o3-mini-high, a compute-intensive and reasoning-focused LLM to conduct structured ROB assessments for RCTs using the ROB 2. The LLM&#x2019;s performance reflected a hybrid cognitive task: first performing semantic reasoning to extract and interpret complex clinical narratives for answering signaling questions and subsequently strictly following instructions by applying the structured logic of the ROB 2 decision tree. The judgments by review authors from published Cochrane systematic reviews served as a widely recognized reference standard for assessing the generalizability and representativeness of the LLM&#x2019;s evaluations.</p><p>The LLM demonstrated a moderate mean accuracy of 74.5%. At the domain level, the averaged sensitivity (57.4%) and specificity (78.4%), calculated from 2 independent assessments, indicated a lower likelihood of identifying high-risk domain-level judgments while reliably detecting low-risk judgments. However, this pattern was not reflected in the overall trial-level assessment. When domain-level judgments were integrated into an overall ROB classification, the model showed an asymmetric pattern in the opposite direction, tending to classify trials as high risk rather than overlook truly high-risk trials. Specifically, across the 2 assessments, the overall judgments resulted in 9 FPs and only 1 FN. This tendency may be advantageous in screening applications by increasing the likelihood that studies with potential bias will be prioritized for subsequent expert review.</p><p>In evaluating performance across trial- and domain-specific contexts, we observed substantial variation in accuracy across medical disciplines, suggesting that the model&#x2019;s alignment with expert judgments may depend on the characteristics of the underlying literature. This variability may partly reflect known challenges in ROB assessment, such as complex study structures, nonintuitive reporting terminology, and inconsistencies even among experienced reviewers, as noted in previous research [<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>]. Our findings provide a preliminary overview of model performance across medical disciplines. While high accuracy was observed in structured, objective end points such as bone health, these results should be interpreted with caution due to the limited number of trials in these subgroups. Accuracy diminished to approximately 60% in disciplines such as psychology [<xref ref-type="bibr" rid="ref55">55</xref>] and cardiology [<xref ref-type="bibr" rid="ref56">56</xref>], suggesting that while the model excels at processing structured data, its reliability diminishes when critical details such as cointerventions or adherence are implicit or conveyed through nuanced narratives. Future implementation should prioritize automated screening for trials with objective outcomes while maintaining intensive expert oversight for behavioral research.</p><p>At the domain-specific level, the model demonstrated the highest accuracy in domain 4 (measurement of the outcome), which presented the fewest discrepancies. Meanwhile, domains with well-defined criteria and clearly stated descriptions, such as domain 1 (randomization process), exhibited the highest &#x03BA; value (0.86) despite a relatively higher frequency of judgment discrepancies compared with the reference standard. In contrast, domain 2 (deviations from intended interventions) exhibited the lowest consistency, with a &#x03BA; value of 0.39 and the lowest observed agreement. This discrepancy likely reflects that the fixed structure of the prompt struggles with the inherent complexity of this domain. Following the ROB 2, our prompts required the LLM to initially categorize the trial as evaluating either an intention-to-treat or per-protocol effect. When trial reports use obscure or nonstandard terminology without further elaboration, the model&#x2019;s initial classification is prone to error, resulting in cascading inconsistencies in the subsequent assessment. This aligns with empirical evidence [<xref ref-type="bibr" rid="ref57">57</xref>] suggesting that human raters also demonstrate the lowest interrater reliability in domain 2, often due to the nuanced interpretive requirements of intention-to-treat and per-protocol effects. These issues present challenges not only for LLMs but also for human reviewers, contributing to discrepancies in judgment and highlighting the need for more precise information extraction, robust reasoning, and contextual understanding in automated ROB assessments.</p><p>A total of 74 discrepancies were observed between LLM assessments and the reference standard, with 55 (74.3%) related to differences in final risk judgments rather than data extraction errors. Most judgment-related discrepancies occurred in domains 1 and 5. Analysis showed that the explanations generated by LLMs often highlighted the presence or absence of specific keywords (eg, &#x201C;concealment&#x201D;) rather than synthesizing indirect cues or context. For instance, in domain 1, the absence of explicit terms such as &#x201C;allocation concealment&#x201D; in the model&#x2019;s reasons was associated with a &#x201C;some concerns&#x201D; rating, whereas human reviewers inferred low risk based on indirect phrases such as &#x201C;centralized randomization.&#x201D; A similar discrepancy was observed in domain 5, where vague reporting often led to inconsistent bias judgments. This suggests that the textual outputs observed in the model&#x2019;s generated justifications remain highly sensitive to surface-level terminology.</p><p>These discrepancies highlight a fundamental distinction between human and LLM-based ROB assessments: humans integrate contextual understanding and inferential logic, whereas LLMs&#x2019; output may appear influenced by surface features. Nevertheless, the model&#x2019;s strong internal consistency and performance in well-structured domains suggest its potential as a tool in systematic review workflows.</p><p>In addition, our study used GPT-o3-mini-high, an LLM optimized for reasoning and instruction-following capabilities that was explicitly selected for its ability to perform inference tasks to conduct a rule-guided assessment task [<xref ref-type="bibr" rid="ref7">7</xref>]. Compared to the models used in the previous study [<xref ref-type="bibr" rid="ref5">5</xref>] (ChatGPT and Claude), which were based on earlier-generation architectures, the GPT-o3-mini model has been shown to generate more accurate and clearer answers by using deeper internal reasoning processes. The GPT-o3-mini model has demonstrated a 39% reduction in major errors on complex questions and was preferred over GPT-o1-mini in 56% of expert tester comparisons [<xref ref-type="bibr" rid="ref7">7</xref>]. Furthermore, its performance on high-level scientific reasoning benchmarks such as GPQA Diamond (79.7%) significantly outperforms contemporary nonreasoning models such as Claude 3.5 Sonnet (65.0%) [<xref ref-type="bibr" rid="ref58">58</xref>]. This demonstrates its superior capacity for tasks requiring nuanced judgment, such as ROB evaluation.</p><p>Finally, the methodological tools differed between studies. Lai et al [<xref ref-type="bibr" rid="ref5">5</xref>] assessed bias using a modified version of the original Cochrane ROB tool developed by the CLARITY group [<xref ref-type="bibr" rid="ref59">59</xref>]. In contrast, our study applied the updated version of the tool, which is currently recommended by the Cochrane Collaboration for evaluating randomized trials [<xref ref-type="bibr" rid="ref3">3</xref>]. Despite its conceptual improvements, prior research has demonstrated that the ROB 2 exhibits relatively poor interrater reliability [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref60">60</xref>]. It involves more steps and requires more logical reasoning to apply consistently, thereby presenting inherent challenges for consistent evaluation. This complexity makes it particularly suitable for testing LLMs with higher reasoning capacity.</p></sec><sec id="s4-2"><title>Limitations</title><p>This study has several limitations that should be noted. First, while Cochrane systematic reviews represent a high standard of methodological rigor [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>], their assessments are not definitive or immune to error. Our comparisons reflect alignment with expert assessments as a reference standard rather than assuming those assessments to be correct. Moreover, the ROB 2 is known to have relatively poor interrater reliability and produce inconsistent judgments even among human raters [<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref60">60</xref>]. This inherent variability likely contributed to the discrepancies between the LLM-generated and expert assessments. Second, we evaluated a single LLM under fixed prompting conditions without exploring performance across different models, prompt styles, or fine-tuning approaches. Although we iteratively refined prompts using a small pilot set of RCTs based on a previous study [<xref ref-type="bibr" rid="ref5">5</xref>] and achieved consistent alignment with experts, the approximately 75% accuracy observed in the validation dataset suggests potential limitations in generalizing prompts optimized on a small pilot test to a broader validation dataset. Third, our dataset, limited to 29 RCTs from 11 Cochrane systematic reviews, may not fully represent the diversity across all medical disciplines. Fourth, a limitation involves the potential for pretraining data contamination. Given the nature of LLMs, it is difficult to definitively distinguish between the retrieval of memorized patterns and de novo reasoning. Although mitigation efforts were made, the possibility of prior exposure to public medical datasets remains a fundamental constraint when evaluating LLM outputs. Finally, this study using the ChatGPT web interface introduced certain methodological fragility. As the precise mechanism of PDF parsing within the interface remains uncertain, the risk of data omission or formatting errors cannot be entirely ruled out, highlighting the need for future validation studies to use version-controlled APIs.</p></sec><sec id="s4-3"><title>Conclusions</title><p>In this exploratory feasibility study using one of the most advanced LLMs available, we found that the accuracy of ROB assessments for randomized trials was largely comparable to that of Cochrane review authors when using the ROB 2. These preliminary findings suggest the potential of LLMs in methodological evaluations, although domains with less structured reporting demand careful consideration. With further development, LLMs could significantly enhance the efficiency and capacity for large-scale application of systematic review processes in biomedical research.</p></sec></sec></body><back><ack><p>During this work, the authors used the GPT-o3-mini-high model (OpenAI) as part of a formal research design, with a clear description provided in the Methods section. After using this tool, the authors reviewed and edited the outputs as needed and take full responsibility for the content of the publication.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed during this study are included in this published article and its supplementary materials.</p></sec></notes><fn-group><fn fn-type="con"><p>YJL contributed to conceptualization, methodology, data curation, investigation, formal analysis, validation, statistical analysis, writing&#x2014;original draft, and visualization. SHL contributed to writing&#x2014;original draft, writing&#x2014;review and editing, and critical revision for important intellectual content. JWL contributed to conceptualization, methodology, supervision, writing&#x2014;review and editing, validation, project administration, and resources.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">FN</term><def><p>false negative</p></def></def-item><def-item><term id="abb2">FP</term><def><p>false positive</p></def></def-item><def-item><term id="abb3">GRADE</term><def><p>Grading of Recommendations Assessment, Development, and Evaluation</p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb5">PABA</term><def><p>prevalence-adjusted, bias-adjusted</p></def></def-item><def-item><term id="abb6">RCT</term><def><p>randomized controlled trial</p></def></def-item><def-item><term id="abb7">RD</term><def><p>relative difference</p></def></def-item><def-item><term id="abb8">ROB</term><def><p>risk of bias</p></def></def-item><def-item><term id="abb9">ROB 2</term><def><p>version 2 of the Cochrane risk-of-bias tool for randomized trials</p></def></def-item><def-item><term id="abb10">TN</term><def><p>true negative</p></def></def-item><def-item><term id="abb11">TP</term><def><p>true positive</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fanaroff</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Califf</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Lopes</surname><given-names>RD</given-names> </name></person-group><article-title>High-quality evidence to inform clinical practice</article-title><source>Lancet</source><year>2019</year><month>08</month><day>24</day><volume>394</volume><issue>10199</issue><fpage>633</fpage><lpage>634</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(19)31256-5</pub-id><pub-id pub-id-type="medline">31448730</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guyatt</surname><given-names>G</given-names> </name><name name-style="western"><surname>Agoritsas</surname><given-names>T</given-names> </name><name name-style="western"><surname>Brignardello-Petersen</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Core GRADE 1: overview of the Core GRADE approach</article-title><source>BMJ</source><year>2025</year><month>04</month><day>22</day><volume>389</volume><fpage>e081903</fpage><pub-id pub-id-type="doi">10.1136/bmj-2024-081903</pub-id><pub-id pub-id-type="medline">40262844</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sterne</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Savovi&#x0107;</surname><given-names>J</given-names> </name><name name-style="western"><surname>Page</surname><given-names>MJ</given-names> </name><etal/></person-group><article-title>RoB 2: a revised tool for assessing risk of bias in randomised trials</article-title><source>BMJ</source><year>2019</year><month>08</month><day>28</day><volume>366</volume><fpage>l4898</fpage><pub-id pub-id-type="doi">10.1136/bmj.l4898</pub-id><pub-id pub-id-type="medline">31462531</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bedi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Orr-Ewing</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Testing and evaluation of health care applications of large language models: a systematic review</article-title><source>JAMA</source><year>2025</year><month>01</month><day>28</day><volume>333</volume><issue>4</issue><fpage>319</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1001/jama.2024.21700</pub-id><pub-id pub-id-type="medline">39405325</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lai</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ge</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Assessing the risk of bias in randomized clinical trials with large language models</article-title><source>JAMA Netw Open</source><year>2024</year><month>05</month><day>1</day><volume>7</volume><issue>5</issue><fpage>e2412687</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.12687</pub-id><pub-id pub-id-type="medline">38776081</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pitt</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Schwartz</surname><given-names>TA</given-names> </name><name name-style="western"><surname>Chu</surname><given-names>D</given-names> </name></person-group><article-title>AAPOR reporting guidelines for survey studies</article-title><source>JAMA Surg</source><year>2021</year><month>08</month><day>1</day><volume>156</volume><issue>8</issue><fpage>785</fpage><lpage>786</lpage><pub-id pub-id-type="doi">10.1001/jamasurg.2021.0543</pub-id><pub-id pub-id-type="medline">33825811</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="web"><article-title>OpenAI o3&#x2011;mini</article-title><source>OpenAI</source><access-date>2025-04-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openai.com/index/openai-o3-mini/">https://openai.com/index/openai-o3-mini/</ext-link></comment></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="web"><article-title>Scope of human research projects exempt from institutional review board review [Article in Chinese]</article-title><source>The Executive Yuan Gazette</source><year>2012</year><access-date>2025-05-25</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://gazette.nat.gov.tw/EG_FileManager/eguploadpub/eg018127/ch08/type1/gov70/num35/Eg.htm">https://gazette.nat.gov.tw/EG_FileManager/eguploadpub/eg018127/ch08/type1/gov70/num35/Eg.htm</ext-link></comment></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="web"><article-title>RoB 2: a revised Cochrane risk-of-bias tool for randomized trials</article-title><source>Cochrane Methods Bias</source><access-date>2025-04-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://methods.cochrane.org/bias/resources/rob-2-revised-cochrane-risk-bias-tool-randomized-trials">https://methods.cochrane.org/bias/resources/rob-2-revised-cochrane-risk-bias-tool-randomized-trials</ext-link></comment></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maher</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Sayyed</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Elkhouly</surname><given-names>NI</given-names> </name></person-group><article-title>Different routes and forms of uterotonics for treatment of retained placenta: a randomized clinical trial</article-title><source>J Matern Fetal Neonatal Med</source><year>2017</year><month>09</month><volume>30</volume><issue>18</issue><fpage>2179</fpage><lpage>2184</lpage><comment>Retracted in</comment><comment>J Matern Fetal Neonatal Med. 2025 Dec;38(1):2509344</comment><pub-id pub-id-type="doi">10.1080/14767058.2025.2509344</pub-id><pub-id pub-id-type="medline">40533268</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><article-title>Statement of retraction: Different routes and forms of uterotonics for treatment of retained placenta: a randomized clinical trial</article-title><source>J Matern Fetal Neonatal Med</source><year>2025</year><month>12</month><volume>38</volume><issue>1</issue><fpage>2509344</fpage><pub-id pub-id-type="doi">10.1080/14767058.2025.2509344</pub-id><pub-id pub-id-type="medline">40533268</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dosenovic</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jelicic Kadic</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vucic</surname><given-names>K</given-names> </name><name name-style="western"><surname>Markovina</surname><given-names>N</given-names> </name><name name-style="western"><surname>Pieper</surname><given-names>D</given-names> </name><name name-style="western"><surname>Puljak</surname><given-names>L</given-names> </name></person-group><article-title>Comparison of methodological quality rating of systematic reviews on neuropathic pain using AMSTAR and R-AMSTAR</article-title><source>BMC Med Res Methodol</source><year>2018</year><month>05</month><day>8</day><volume>18</volume><fpage>37</fpage><pub-id pub-id-type="doi">10.1186/s12874-018-0493-y</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Methodological and reporting quality in non-Cochrane systematic review updates could be improved: a comparative study</article-title><source>J Clin Epidemiol</source><year>2020</year><month>03</month><volume>119</volume><fpage>36</fpage><lpage>46</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2019.11.012</pub-id><pub-id pub-id-type="medline">31759063</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marenzi</surname><given-names>G</given-names> </name><name name-style="western"><surname>Muratori</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cosentino</surname><given-names>ER</given-names> </name><etal/></person-group><article-title>Continuous ultrafiltration for congestive heart failure: the CUORE trial</article-title><source>J Card Fail</source><year>2014</year><month>01</month><volume>20</volume><issue>1</issue><fpage>9</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1016/j.cardfail.2013.11.004</pub-id><pub-id pub-id-type="medline">24269855</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Li</surname><given-names>P</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>Effect of pretreatment of S-ketamine on postoperative depression for breast cancer patients</article-title><source>J Invest Surg</source><year>2021</year><month>08</month><volume>34</volume><issue>8</issue><fpage>883</fpage><lpage>888</lpage><pub-id pub-id-type="doi">10.1080/08941939.2019.1710626</pub-id><pub-id pub-id-type="medline">31948296</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Katial</surname><given-names>RK</given-names> </name><name name-style="western"><surname>Bernstein</surname><given-names>D</given-names> </name><name name-style="western"><surname>Prazma</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Lincourt</surname><given-names>WR</given-names> </name><name name-style="western"><surname>Stempel</surname><given-names>DA</given-names> </name></person-group><article-title>Long-term treatment with fluticasone propionate/salmeterol via Diskus improves asthma control versus fluticasone propionate alone</article-title><source>Allergy Asthma Proc</source><year>2011</year><volume>32</volume><issue>2</issue><fpage>127</fpage><lpage>136</lpage><pub-id pub-id-type="doi">10.2500/aap.2011.32.3426</pub-id><pub-id pub-id-type="medline">21189151</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Munteanu</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Landorf</surname><given-names>KB</given-names> </name><name name-style="western"><surname>McClelland</surname><given-names>JA</given-names> </name><etal/></person-group><article-title>Shoe-stiffening inserts for first metatarsophalangeal joint osteoarthritis: a randomised trial</article-title><source>Osteoarthritis Cartilage</source><year>2021</year><month>04</month><volume>29</volume><issue>4</issue><fpage>480</fpage><lpage>490</lpage><pub-id pub-id-type="doi">10.1016/j.joca.2021.02.002</pub-id><pub-id pub-id-type="medline">33588086</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lebares</surname><given-names>CC</given-names> </name><name name-style="western"><surname>Hershberger</surname><given-names>AO</given-names> </name><name name-style="western"><surname>Guvva</surname><given-names>EV</given-names> </name><etal/></person-group><article-title>Feasibility of formal mindfulness-based stress-resilience training among surgery interns: a randomized clinical trial</article-title><source>JAMA Surg</source><year>2018</year><month>10</month><day>1</day><volume>153</volume><issue>10</issue><fpage>e182734</fpage><pub-id pub-id-type="doi">10.1001/jamasurg.2018.2734</pub-id><pub-id pub-id-type="medline">30167655</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Daher</surname><given-names>EF</given-names> </name><name name-style="western"><surname>Nogueira</surname><given-names>CB</given-names> </name></person-group><article-title>Evaluation of penicillin therapy in patients with leptospirosis and acute renal failure</article-title><source>Rev Inst Med Trop Sao Paulo</source><year>2000</year><volume>42</volume><issue>6</issue><fpage>327</fpage><lpage>332</lpage><pub-id pub-id-type="doi">10.1590/s0036-46652000000600005</pub-id><pub-id pub-id-type="medline">11136519</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Toouli</surname><given-names>J</given-names> </name><name name-style="western"><surname>Roberts-Thomson</surname><given-names>IC</given-names> </name><name name-style="western"><surname>Kellow</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Manometry based randomised trial of endoscopic sphincterotomy for sphincter of Oddi dysfunction</article-title><source>Gut</source><year>2000</year><month>01</month><volume>46</volume><issue>1</issue><fpage>98</fpage><lpage>102</lpage><pub-id pub-id-type="doi">10.1136/gut.46.1.98</pub-id><pub-id pub-id-type="medline">10601063</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hanna</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>WH</given-names> </name><name name-style="western"><surname>Teo</surname><given-names>BW</given-names> </name><etal/></person-group><article-title>Extracorporeal ultrafiltration vs. conventional diuretic therapy in advanced decompensated heart failure</article-title><source>Congest Heart Fail</source><year>2012</year><volume>18</volume><issue>1</issue><fpage>54</fpage><lpage>63</lpage><pub-id pub-id-type="doi">10.1111/j.1751-7133.2011.00231.x</pub-id><pub-id pub-id-type="medline">22277179</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Efficacy and safety of early ultrafiltration in patients with acute decompensated heart failure with volume overload: a prospective, randomized, controlled clinical trial</article-title><source>BMC Cardiovasc Disord</source><year>2020</year><month>10</month><day>14</day><volume>20</volume><issue>1</issue><fpage>447</fpage><pub-id pub-id-type="doi">10.1186/s12872-020-01733-5</pub-id><pub-id pub-id-type="medline">33054727</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tavakoli Ardakani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mehrpooya</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mehdizadeh</surname><given-names>M</given-names> </name><name name-style="western"><surname>Beiraghi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hajifathali</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kazemi</surname><given-names>MH</given-names> </name></person-group><article-title>Sertraline treatment decreased the serum levels of interleukin-6 and high-sensitivity C-reactive protein in hematopoietic stem cell transplantation patients with depression; a randomized double-blind, placebo-controlled clinical trial</article-title><source>Bone Marrow Transplant</source><year>2020</year><month>04</month><volume>55</volume><issue>4</issue><fpage>830</fpage><lpage>832</lpage><pub-id pub-id-type="doi">10.1038/s41409-019-0623-0</pub-id><pub-id pub-id-type="medline">31383995</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cavalcanti</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Zampieri</surname><given-names>FG</given-names> </name><name name-style="western"><surname>Rosa</surname><given-names>RG</given-names> </name><etal/></person-group><article-title>Hydroxychloroquine with or without azithromycin in mild-to-moderate Covid-19</article-title><source>N Engl J Med</source><year>2020</year><month>11</month><day>19</day><volume>383</volume><issue>21</issue><fpage>2041</fpage><lpage>2052</lpage><pub-id pub-id-type="doi">10.1056/NEJMoa2019014</pub-id><pub-id pub-id-type="medline">32706953</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peters</surname><given-names>SP</given-names> </name><name name-style="western"><surname>Bleecker</surname><given-names>ER</given-names> </name><name name-style="western"><surname>Canonica</surname><given-names>GW</given-names> </name><etal/></person-group><article-title>Serious asthma events with budesonide plus formoterol vs. budesonide alone</article-title><source>N Engl J Med</source><year>2016</year><month>09</month><day>1</day><volume>375</volume><issue>9</issue><fpage>850</fpage><lpage>860</lpage><pub-id pub-id-type="doi">10.1056/NEJMoa1511190</pub-id><pub-id pub-id-type="medline">27579635</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dami&#x00E3;o Neto</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lucchetti</surname><given-names>AL</given-names> </name><name name-style="western"><surname>da Silva Ezequiel</surname><given-names>O</given-names> </name><name name-style="western"><surname>Lucchetti</surname><given-names>G</given-names> </name></person-group><article-title>Effects of a required large-group mindfulness meditation course on first-year medical students&#x2019; mental health and quality of life: a randomized controlled trial</article-title><source>J Gen Intern Med</source><year>2020</year><month>03</month><volume>35</volume><issue>3</issue><fpage>672</fpage><lpage>678</lpage><pub-id pub-id-type="doi">10.1007/s11606-019-05284-0</pub-id><pub-id pub-id-type="medline">31452038</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Omrani</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Pathan</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>SA</given-names> </name><etal/></person-group><article-title>Randomized double-blinded placebo-controlled trial of hydroxychloroquine with or without azithromycin for virologic cure of non-severe Covid-19</article-title><source>EClinicalMedicine</source><year>2020</year><month>12</month><volume>29</volume><fpage>100645</fpage><pub-id pub-id-type="doi">10.1016/j.eclinm.2020.100645</pub-id><pub-id pub-id-type="medline">33251500</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Erogul</surname><given-names>M</given-names> </name><name name-style="western"><surname>Singer</surname><given-names>G</given-names> </name><name name-style="western"><surname>McIntyre</surname><given-names>T</given-names> </name><name name-style="western"><surname>Stefanov</surname><given-names>DG</given-names> </name></person-group><article-title>Abridged mindfulness intervention to support wellness in first-year medical students</article-title><source>Teach Learn Med</source><year>2014</year><volume>26</volume><issue>4</issue><fpage>350</fpage><lpage>356</lpage><pub-id pub-id-type="doi">10.1080/10401334.2014.945025</pub-id><pub-id pub-id-type="medline">25318029</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sekhavati</surname><given-names>E</given-names> </name><name name-style="western"><surname>Jafari</surname><given-names>F</given-names> </name><name name-style="western"><surname>SeyedAlinaghi</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Safety and effectiveness of azithromycin in patients with COVID-19: an open-label randomised trial</article-title><source>Int J Antimicrob Agents</source><year>2020</year><month>10</month><volume>56</volume><issue>4</issue><fpage>106143</fpage><pub-id pub-id-type="doi">10.1016/j.ijantimicag.2020.106143</pub-id><pub-id pub-id-type="medline">32853672</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Total laparoscopic uncut Roux-en-Y for radical distal gastrectomy: an interim analysis of a randomized, controlled, clinical trial</article-title><source>Ann Surg Oncol</source><year>2021</year><month>01</month><volume>28</volume><issue>1</issue><fpage>90</fpage><lpage>96</lpage><pub-id pub-id-type="doi">10.1245/s10434-020-08710-4</pub-id><pub-id pub-id-type="medline">32556870</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van Heeringen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zivkov</surname><given-names>M</given-names> </name></person-group><article-title>Pharmacological treatment of depression in cancer patients. A placebo-controlled study of mianserin</article-title><source>Br J Psychiatry</source><year>1996</year><month>10</month><volume>169</volume><issue>4</issue><fpage>440</fpage><lpage>443</lpage><pub-id pub-id-type="doi">10.1192/bjp.169.4.440</pub-id><pub-id pub-id-type="medline">8894194</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bart</surname><given-names>BA</given-names> </name><name name-style="western"><surname>Boyle</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bank</surname><given-names>AJ</given-names> </name><etal/></person-group><article-title>Ultrafiltration versus usual care for hospitalized patients with heart failure: the Relief for Acutely Fluid-Overloaded Patients With Decompensated Congestive Heart Failure (RAPID-CHF) trial</article-title><source>J Am Coll Cardiol</source><year>2005</year><month>12</month><day>6</day><volume>46</volume><issue>11</issue><fpage>2043</fpage><lpage>2046</lpage><pub-id pub-id-type="doi">10.1016/j.jacc.2005.05.098</pub-id><pub-id pub-id-type="medline">16325039</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>RECOVERY Collaborative Group</collab></person-group><article-title>Azithromycin in patients admitted to hospital with COVID-19 (RECOVERY): a randomised, controlled, open-label, platform trial</article-title><source>Lancet</source><year>2021</year><month>02</month><volume>397</volume><issue>10274</issue><fpage>605</fpage><lpage>612</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(21)00149-5</pub-id><pub-id pub-id-type="medline">33545096</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van Stralen</surname><given-names>G</given-names> </name><name name-style="western"><surname>Veenhof</surname><given-names>M</given-names> </name><name name-style="western"><surname>Holleboom</surname><given-names>C</given-names> </name><name name-style="western"><surname>van Roosmalen</surname><given-names>J</given-names> </name></person-group><article-title>No reduction of manual removal after misoprostol for retained placenta: a double-blind, randomized trial</article-title><source>Acta Obstet Gynecol Scand</source><year>2013</year><month>04</month><volume>92</volume><issue>4</issue><fpage>398</fpage><lpage>403</lpage><pub-id pub-id-type="doi">10.1111/aogs.12065</pub-id><pub-id pub-id-type="medline">23231499</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Takezawa</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kida</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kida</surname><given-names>M</given-names> </name><name name-style="western"><surname>Saigenji</surname><given-names>K</given-names> </name></person-group><article-title>Influence of endoscopic papillary balloon dilation and endoscopic sphincterotomy on sphincter of Oddi function: a randomized controlled trial</article-title><source>Endoscopy</source><year>2004</year><month>07</month><volume>36</volume><issue>7</issue><fpage>631</fpage><lpage>637</lpage><pub-id pub-id-type="doi">10.1055/s-2004-814538</pub-id><pub-id pub-id-type="medline">15243887</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Woodcock</surname><given-names>A</given-names> </name><name name-style="western"><surname>L&#x00F6;tvall</surname><given-names>J</given-names> </name><name name-style="western"><surname>Busse</surname><given-names>WW</given-names> </name><etal/></person-group><article-title>Efficacy and safety of fluticasone furoate 100 &#x03BC;g and 200 &#x03BC;g once daily in the treatment of moderate-severe asthma in adults and adolescents: a 24-week randomised study</article-title><source>BMC Pulm Med</source><year>2014</year><month>07</month><day>9</day><volume>14</volume><fpage>113</fpage><pub-id pub-id-type="doi">10.1186/1471-2466-14-113</pub-id><pub-id pub-id-type="medline">25007865</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>&#x015E;eker</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kayata&#x015F;</surname><given-names>M</given-names> </name><name name-style="western"><surname>H&#x00FC;zmeli</surname><given-names>C</given-names> </name><name name-style="western"><surname>Candan</surname><given-names>F</given-names> </name><name name-style="western"><surname>Y&#x0131;lmaz</surname><given-names>MB</given-names> </name></person-group><article-title>Comparison of ultrafiltration and intravenous diuretic therapies in patients hospitalized for acute decompensated biventricular heart failure</article-title><source>Turk J Nephrol</source><year>2019</year><month>01</month><access-date>2026-08-28</access-date><volume>25</volume><issue>1</issue><fpage>79</fpage><lpage>87</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.turkjnephrol.org/index.php/pub/article/view/1036">https://www.turkjnephrol.org/index.php/pub/article/view/1036</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Bailes</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Barnes</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Efficacy and safety of once-daily single-inhaler triple therapy (FF/UMEC/VI) versus FF/VI in patients with inadequately controlled asthma (CAPTAIN): a double-blind, randomised, phase 3A trial</article-title><source>Lancet Respir Med</source><year>2021</year><month>01</month><volume>9</volume><issue>1</issue><fpage>69</fpage><lpage>84</lpage><pub-id pub-id-type="doi">10.1016/S2213-2600(20)30389-1</pub-id><pub-id pub-id-type="medline">32918892</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schreiber</surname><given-names>S</given-names> </name><name name-style="western"><surname>Khaliq-Kareemi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lawrance</surname><given-names>IC</given-names> </name><etal/></person-group><article-title>Maintenance therapy with certolizumab pegol for Crohn&#x2019;s disease</article-title><source>N Engl J Med</source><year>2007</year><month>07</month><day>19</day><volume>357</volume><issue>3</issue><fpage>239</fpage><lpage>250</lpage><pub-id pub-id-type="doi">10.1056/NEJMoa062897</pub-id><pub-id pub-id-type="medline">17634459</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hinks</surname><given-names>TS</given-names> </name><name name-style="western"><surname>Barber</surname><given-names>VS</given-names> </name><name name-style="western"><surname>Black</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A multi-centre open-label two-arm randomised superiority clinical trial of azithromycin versus usual care in ambulatory COVID-19: study protocol for the ATOMIC2 trial</article-title><source>Trials</source><year>2020</year><month>08</month><day>17</day><volume>21</volume><issue>1</issue><fpage>718</fpage><pub-id pub-id-type="doi">10.1186/s13063-020-04593-8</pub-id><pub-id pub-id-type="medline">32807209</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cotton</surname><given-names>PB</given-names> </name><name name-style="western"><surname>Durkalski</surname><given-names>V</given-names> </name><name name-style="western"><surname>Romagnuolo</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Effect of endoscopic sphincterotomy for suspected sphincter of Oddi dysfunction on pain-related disability following cholecystectomy: the EPISOD randomized clinical trial</article-title><source>JAMA</source><year>2014</year><month>05</month><volume>311</volume><issue>20</issue><fpage>2101</fpage><lpage>2109</lpage><pub-id pub-id-type="doi">10.1001/jama.2014.5220</pub-id><pub-id pub-id-type="medline">24867013</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Srivastava</surname><given-names>M</given-names> </name><name name-style="western"><surname>Harrison</surname><given-names>N</given-names> </name><name name-style="western"><surname>Caetano</surname><given-names>AF</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Law</surname><given-names>M</given-names> </name></person-group><article-title>Ultrafiltration for acute heart failure</article-title><source>Cochrane Database Syst Rev</source><year>2022</year><month>01</month><day>21</day><volume>1</volume><issue>1</issue><fpage>CD013593</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD013593.pub2</pub-id><pub-id pub-id-type="medline">35061249</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vita</surname><given-names>G</given-names> </name><name name-style="western"><surname>Compri</surname><given-names>B</given-names> </name><name name-style="western"><surname>Matcham</surname><given-names>F</given-names> </name><name name-style="western"><surname>Barbui</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ostuzzi</surname><given-names>G</given-names> </name></person-group><article-title>Antidepressants for the treatment of depression in people with cancer</article-title><source>Cochrane Database Syst Rev</source><year>2023</year><month>03</month><day>31</day><volume>3</volume><issue>3</issue><fpage>CD011006</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD011006.pub4</pub-id><pub-id pub-id-type="medline">36999619</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Oba</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Anwer</surname><given-names>S</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>T</given-names> </name><name name-style="western"><surname>Maduke</surname><given-names>T</given-names> </name><name name-style="western"><surname>Dias</surname><given-names>S</given-names> </name></person-group><article-title>Addition of long-acting beta2 agonists or long-acting muscarinic antagonists versus doubling the dose of inhaled corticosteroids (ICS) in adolescents and adults with uncontrolled asthma with medium dose ICS: a systematic review and network meta-analysis</article-title><source>Cochrane Database Syst Rev</source><year>2023</year><month>08</month><day>21</day><volume>8</volume><issue>8</issue><fpage>CD013797</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD013797.pub2</pub-id><pub-id pub-id-type="medline">37602534</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Munteanu</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Buldt</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lithgow</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Cotchett</surname><given-names>M</given-names> </name><name name-style="western"><surname>Landorf</surname><given-names>KB</given-names> </name><name name-style="western"><surname>Menz</surname><given-names>HB</given-names> </name></person-group><article-title>Non-surgical interventions for treating osteoarthritis of the big toe joint</article-title><source>Cochrane Database Syst Rev</source><year>2024</year><month>06</month><day>17</day><volume>6</volume><issue>6</issue><fpage>CD007809</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD007809.pub3</pub-id><pub-id pub-id-type="medline">38884172</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sekhar</surname><given-names>P</given-names> </name><name name-style="western"><surname>Tee</surname><given-names>QX</given-names> </name><name name-style="western"><surname>Ashraf</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Mindfulness-based psychological interventions for improving mental well-being in medical students and junior doctors</article-title><source>Cochrane Database Syst Rev</source><year>2021</year><month>12</month><day>10</day><volume>12</volume><issue>12</issue><fpage>CD013740</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD013740.pub2</pub-id><pub-id pub-id-type="medline">34890044</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sothornwit</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ngamjarus</surname><given-names>C</given-names> </name><name name-style="western"><surname>Pattanittum</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Uterotonics for management of retained placenta</article-title><source>Cochrane Database Syst Rev</source><year>2024</year><month>10</month><day>28</day><volume>10</volume><issue>10</issue><fpage>CD016147</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD016147</pub-id><pub-id pub-id-type="medline">39465684</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Win</surname><given-names>TZ</given-names> </name><name name-style="western"><surname>Han</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Edwards</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Antibiotics for treatment of leptospirosis</article-title><source>Cochrane Database Syst Rev</source><year>2024</year><month>03</month><day>14</day><volume>3</volume><issue>3</issue><fpage>CD014960</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD014960.pub2</pub-id><pub-id pub-id-type="medline">38483092</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naing</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ni</surname><given-names>H</given-names> </name><name name-style="western"><surname>Aung</surname><given-names>HH</given-names> </name><name name-style="western"><surname>Pavlov</surname><given-names>CS</given-names> </name></person-group><article-title>Endoscopic sphincterotomy for adults with biliary sphincter of Oddi dysfunction</article-title><source>Cochrane Database Syst Rev</source><year>2024</year><month>03</month><day>22</day><volume>3</volume><issue>3</issue><fpage>CD014944</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD014944.pub2</pub-id><pub-id pub-id-type="medline">38517086</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Popp</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stegemann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Riemer</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Antibiotics for the treatment of COVID-19</article-title><source>Cochrane Database Syst Rev</source><year>2021</year><month>10</month><day>22</day><volume>10</volume><issue>10</issue><fpage>CD015025</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD015025</pub-id><pub-id pub-id-type="medline">34679203</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cai</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Mu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>Uncut Roux-en-Y reconstruction after distal gastrectomy for gastric cancer</article-title><source>Cochrane Database Syst Rev</source><year>2024</year><month>02</month><day>29</day><volume>2</volume><issue>2</issue><fpage>CD015014</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD015014.pub2</pub-id><pub-id pub-id-type="medline">38421211</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Okabayashi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yamazaki</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yamamoto</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Certolizumab pegol for maintenance of medically induced remission in Crohn&#x2019;s disease</article-title><source>Cochrane Database Syst Rev</source><year>2022</year><month>06</month><day>30</day><volume>6</volume><issue>6</issue><fpage>CD013747</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD013747.pub2</pub-id><pub-id pub-id-type="medline">35771590</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jordan</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Lensen</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Farquhar</surname><given-names>CM</given-names> </name></person-group><article-title>There were large discrepancies in risk of bias tool judgments when a randomized controlled trial appeared in more than one systematic review</article-title><source>J Clin Epidemiol</source><year>2017</year><month>01</month><volume>81</volume><fpage>72</fpage><lpage>76</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2016.08.012</pub-id><pub-id pub-id-type="medline">27622779</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Minozzi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cinquini</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gianola</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gonzalez-Lorenzo</surname><given-names>M</given-names> </name><name name-style="western"><surname>Banzi</surname><given-names>R</given-names> </name></person-group><article-title>The revised Cochrane risk of bias tool for randomized trials (RoB 2) showed low interrater reliability and challenges in its application</article-title><source>J Clin Epidemiol</source><year>2020</year><month>10</month><volume>126</volume><fpage>37</fpage><lpage>44</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2020.06.015</pub-id><pub-id pub-id-type="medline">32562833</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Button</surname><given-names>KS</given-names> </name><name name-style="western"><surname>Munaf&#x00F2;</surname><given-names>MR</given-names> </name></person-group><article-title>Addressing risk of bias in trials of cognitive behavioral therapy</article-title><source>Shanghai Arch Psychiatry</source><year>2015</year><month>06</month><day>25</day><volume>27</volume><issue>3</issue><fpage>144</fpage><lpage>148</lpage><pub-id pub-id-type="doi">10.11919/j.issn.1002-0829.215042</pub-id><pub-id pub-id-type="medline">26300596</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baasan</surname><given-names>O</given-names> </name><name name-style="western"><surname>Freihat</surname><given-names>O</given-names> </name><name name-style="western"><surname>Nagy</surname><given-names>DU</given-names> </name><name name-style="western"><surname>Lohner</surname><given-names>S</given-names> </name></person-group><article-title>Methodological quality and risk of bias assessment of cardiovascular disease research: analysis of randomized controlled trials published in 2017</article-title><source>Front Cardiovasc Med</source><year>2022</year><volume>9</volume><fpage>830070</fpage><pub-id pub-id-type="doi">10.3389/fcvm.2022.830070</pub-id><pub-id pub-id-type="medline">35369336</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Pitre</surname><given-names>T</given-names> </name><name name-style="western"><surname>Jassal</surname><given-names>T</given-names> </name><name name-style="western"><surname>Talukdar</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Shahab</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ling</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zeraatkar</surname><given-names>D</given-names> </name></person-group><article-title>ChatGPT for assessing risk of bias of randomized trials using the RoB 2.0 tool: a methods study</article-title><source>medRxiv</source><comment>Preprint posted online on  Nov 22, 2023</comment><pub-id pub-id-type="doi">10.1101/2023.11.19.23298727</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="web"><article-title>Claude 3.7 Sonnet and Claude Code</article-title><source>Anthropic</source><year>2025</year><access-date>2025-05-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.anthropic.com/news/claude-3-7-sonnet">https://www.anthropic.com/news/claude-3-7-sonnet</ext-link></comment></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="web"><article-title>Tool to assess risk of bias in randomized controlled trials</article-title><source>DistillerSR</source><access-date>2025-05-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.distillersr.com/resources/methodological-resources/tool-to-assess-risk-of-bias-in-randomized-controlled-trials-distillersr">https://www.distillersr.com/resources/methodological-resources/tool-to-assess-risk-of-bias-in-randomized-controlled-trials-distillersr</ext-link></comment></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eisele-Metzger</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lieberum</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Toews</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Exploring the potential of Claude 2 for risk of bias assessment: using a large language model to assess randomized controlled trials with RoB 2</article-title><source>Res Synth Methods</source><year>2025</year><month>05</month><volume>16</volume><issue>3</issue><fpage>491</fpage><lpage>508</lpage><pub-id pub-id-type="doi">10.1017/rsm.2025.12</pub-id><pub-id pub-id-type="medline">41626932</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Prompt for the large language model to assess risk of bias (ROB) in randomized controlled trials using version 2 of the Cochrane ROB tool for randomized trials.</p><media xlink:href="jmir_v28i1e84915_app1.pdf" xlink:title="PDF File, 205 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Response from ChatGPT.</p><media xlink:href="jmir_v28i1e84915_app2.pdf" xlink:title="PDF File, 1689 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Supplementary tables for reference standard, interpretation criteria, and assessment results.</p><media xlink:href="jmir_v28i1e84915_app3.pdf" xlink:title="PDF File, 1594 KB"/></supplementary-material></app-group></back></article>