<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e90693</article-id><article-id pub-id-type="doi">10.2196/90693</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Model and Task-Aware Test-Time Scaling Strategies for Large Language and Vision-Language Models in Medicine: Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Oh</surname><given-names>Gyutaek</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kim</surname><given-names>Seoyeon</given-names></name><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Park</surname><given-names>Sangjoon</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name name-style="western"><surname>Kim</surname><given-names>Byung-Hoon</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Biomedical Systems Informatics, College of Medicine, Yonsei University</institution><addr-line>50-1 Yonsei-ro, Seodaemun-gu</addr-line><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff2"><institution>Yonsei Institute for Digital Health</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff3"><institution>College of Medicine, Yonsei University</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff4"><institution>Department of Radiation Oncology, College of Medicine, Yonsei University</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff5"><institution>Heavy Ion Therapy Research Institute, Yonsei University College of Medicine</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff6"><institution>Yonsei Cancer Center, Yonsei University College of Medicine</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff7"><institution>Department of Psychiatry, College of Medicine, Yonsei University</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff8"><institution>Institute of Behavioral Sciences in Medicine, Yonsei University College of Medicine</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Ni</surname><given-names>Jun</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Kaya</surname><given-names>Mehmet Onurcan</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Zhang</surname><given-names>Tianyu</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Byung-Hoon Kim, MD, PhD, Department of Biomedical Systems Informatics, College of Medicine, Yonsei University, 50-1 Yonsei-ro, Seodaemun-gu, Seoul, 03722, Republic of Korea, 82 2-2228-2484; <email>egyptdj@yonsei.ac.kr</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>23</day><month>7</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e90693</elocation-id><history><date date-type="received"><day>03</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>06</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>19</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Gyutaek Oh, Seoyeon Kim, Sangjoon Park, Byung-Hoon Kim. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 23.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e90693"/><abstract><sec><title>Background</title><p>Test-time scaling has emerged as a promising method to enhance the reasoning capabilities of large language models (LLMs) and vision-language models (VLMs) during inference without additional training. While foundational studies established scaling paradigms in general domains, their applicability to the unique complexities of medical AI remains underexplored.</p></sec><sec><title>Objective</title><p>This study aims to conduct a comprehensive investigation of test-time scaling in the medical domain. We evaluate the impact of scaling across different model sizes and task complexities. Furthermore, we seek to identify domain-specific bottlenecks and assess model robustness against user-driven perturbations, such as misleading clinical authority.</p></sec><sec sec-type="methods"><title>Methods</title><p>This study evaluated a diverse set of general and medical-specific LLMs and VLMs. Experiments used five textual medical benchmarks comprising over 5500 questions and two multimodal benchmarks comprising 7000 samples. Performance was measured under three scaling conditions: increasing token budgets, iterative sequential scaling, and parallel scaling. Robustness was tested by embedding misleading hints with varying tones and levels of simulated clinical expertise into prompts.</p></sec><sec sec-type="results"><title>Results</title><p>For nonreasoning LLMs, accuracy saturated quickly, with token usage often remaining under 500 tokens regardless of budget increases. Reasoning models demonstrated significant performance gains on complex tasks as token budgets increased. Notably, we identified distinct domain-specific behaviors. First, current VLMs showed a structural bottleneck in integrating visual clues and experienced limited benefit from token expansion. Second, medically fine-tuned LLMs excelled in clinical question answering but exhibited degraded scaling efficiency on calculation tasks compared to general-domain models. This reflects a disparity between qualitative clinical alignment and procedural logic. Third, while optimal scaling improved robustness, models exhibited a cognitive vulnerability by readily abandoning correct reasoning when confronted with misleading expert physician hints. Regarding scaling strategies, parallel scaling outperformed sequential scaling on easier tasks. Conversely, extended sequential scaling or increased budgets proved essential for complex problem-solving.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Test-time scaling rules from general domains do not perfectly translate to medical AI. Longer reasoning is not universally beneficial. Concise reasoning with parallel scaling is optimal for simpler tasks. An extended chain of thought via sequential scaling or increased budgets is required for complex problems. Furthermore, safe clinical deployment requires addressing fundamental vision-language alignment, balancing clinical and procedural reasoning, and mitigating vulnerabilities to perceived clinical authority.</p></sec></abstract><kwd-group><kwd>test-time scaling</kwd><kwd>large language models</kwd><kwd>vision-language models</kwd><kwd>reasoning models</kwd><kwd>medical AI</kwd><kwd>clincal reasoning</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>In recent years, large language models (LLMs) have undergone rapid development.</p><p>Since the introduction of OpenAI&#x2019;s GPT-3 [<xref ref-type="bibr" rid="ref1">1</xref>], both industry and academia have actively competed to develop powerful LLMs [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref9">9</xref>], which are now widely adopted in daily life. More recently, models that integrate data from multiple modalities, particularly vision-language models (VLMs) [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref16">16</xref>], have garnered significant attention among multimodal approaches.</p><p>A common trend in developing these models has been to scale up both model size and training data in pursuit of improved performance. However, the high demand for computational resources and massive datasets presents substantial barriers to broader accessibility and development. Moreover, while training-time scaling laws have led to certain improvements, many models still struggle with complex reasoning tasks.</p><p>To address these challenges, test-time scaling has recently emerged as an effective method for enhancing LLM performance during inference [<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref20">20</xref>]. For instance, rather than generating an immediate final prediction, a model can be configured to produce intermediate step-by-step logical deductions or generate multiple candidate responses to find a reliable consensus. Without requiring additional training or fine-tuning, test-time scaling improves both reliability and accuracy, especially on tasks that require multistep reasoning. By increasing the token budget or generating multiple candidate responses, test-time scaling enables models to produce more detailed and structured chain-of-thought reasoning. This approach is particularly synergistic with reasoning-optimized models, which are trained using supervised fine-tuning (SFT) on chain-of-thought&#x2013;annotated datasets or reinforcement learning (RL) with specialized objectives [<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref23">23</xref>]. The combination of reasoning models and test-time scaling has shown state-of-the-art results on complex tasks such as mathematics problem solving and program code synthesis [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>Recently, applications of LLMs and VLMs in the medical domain have also gained momentum [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref30">30</xref>]. Alongside these applications, both reasoning models [<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref33">33</xref>] and test-time scaling [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>] are being increasingly adopted in medical AI research. Recent studies demonstrate that the use of reasoning models and test-time scaling significantly boosts performance on medical benchmark datasets [<xref ref-type="bibr" rid="ref30">30</xref>].</p><p>Despite growing interest in test-time scaling in the medical domain, many important aspects remain underexplored. Although various test-time scaling strategies have been proposed, only a limited subset has been systematically evaluated in medical applications. For example, it remains unclear whether shorter or longer reasoning is more advantageous in the medical domain, or whether sequential or parallel scaling is more effective. Furthermore, despite differences in training data and methods across models, many prior studies have applied test-time scaling strategies uniformly, without accounting for each model&#x2019;s unique characteristics. Given the wide range of difficulty in medical tasks, it is also important to examine whether reasoning via test-time scaling is equally beneficial across different task types. In addition, with the increasing relevance of VLMs in clinical and diagnostic contexts, investigating test-time scaling strategies for VLMs in the medical domain is essential.</p><p>Crucially, while foundational studies have established test-time scaling paradigms in the general domain by focusing primarily on text-based logic such as mathematics and coding [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>], their findings cannot readily predict the multilayered complexities inherent to clinical AI. General-domain frameworks fail to capture the domain-specific bottlenecks and unique behavioral shifts that emerge in medical environments. Specifically, it remains unexamined how test-time compute interacts with the intricate visual reasoning required for complex medical images, whether specialized medical knowledge fine-tuning paradoxically compromises a model&#x2019;s baseline numerical reasoning in clinical calculations, and how scaling affects a model&#x2019;s cognitive vulnerability when confronted with incorrect information embedded in user prompts. Resolving these questions is vital to discovering insights unique to medicine that transcend mere empirical replication of general-domain findings.</p><p>To bridge these gaps, in this paper, we present a comprehensive investigation of test-time scaling for medical applications. <xref ref-type="fig" rid="figure1">Figure 1</xref> illustrates the overall framework of our study. First, we analyze how increasing the token budget affects the performance of various LLMs and VLMs across multiple medical benchmark datasets. We explore how this performance varies with factors such as model size, model characteristics, and task complexity. Next, we compare sequential and parallel test-time scaling strategies, highlighting their relative effectiveness in medical tasks. Finally, we evaluate the robustness of test-time scaling under user-driven factors, such as misleading contextual information embedded in prompts. Ultimately, the primary aim of this study is to establish a domain-specific framework for scaling inference compute in medical AI. We hypothesize that the optimal scaling strategy is fundamentally dependent on both the inherent reasoning capacity of the model and the specific complexity of the clinical task, and that appropriate scaling can significantly improve model robustness against misleading user inputs.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Overview of our study: investigating test-time scaling for LLMs and VLMs in medicine. LLM: large language model; VLM: vision-language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90693_fig01.png"/></fig></sec><sec id="s1-2"><title>Prior Work</title><sec id="s1-2-1"><title>Test-Time Scaling</title><p>Test-time scaling has recently emerged as a method for enhancing the reasoning capabilities of LLMs during inference by increasing the number of tokens used for chain-of-thought reasoning. Following the introduction of OpenAI&#x2019;s o1 model [<xref ref-type="bibr" rid="ref7">7</xref>], which demonstrated strong performance on complex tasks through enhanced chain-of-thought prompting, many subsequent studies have begun to explore test-time scaling as a means of improving LLM performance without additional training.</p><p>The s1 model [<xref ref-type="bibr" rid="ref17">17</xref>] introduces test-time scaling through &#x201C;budget forcing,&#x201D; controlling computational effort during inference by appending &#x201C;wait&#x201D; tokens to extend thinking or forcefully terminating the response. Remarkably, s1 uses only 1000 curated questions with reasoning paths selected for difficulty, diversity, and quality, yet exceeds the o1-preview model on competition mathematics questions by up to 27%. On the other hand, subsequent studies [<xref ref-type="bibr" rid="ref20">20</xref>] revealed that longer chain-of-thought responses do not consistently enhance accuracy, where correct solutions are often shorter than incorrect ones. This phenomenon suggests that excessive self-revision may degrade model performance rather than enhance it.</p><p>In response to these findings, recent studies have proposed test-time scaling strategies that either extend chain-of-thought reasoning sequentially or generate multiple candidate responses in parallel, aiming to discover optimal configurations that balance depth and diversity of reasoning. In the study by Balachandran et al [<xref ref-type="bibr" rid="ref36">36</xref>], it was demonstrated that both sequential and parallel scaling can improve the performance of nonreasoning and reasoning models. For instance, parallel scaling significantly boosted accuracy on the Traveling Salesman Problem easy subset, improving performance from 42% to 95% when scaled to 256 parallel API calls. More broadly, Snell et al [<xref ref-type="bibr" rid="ref18">18</xref>] emphasized that identifying the optimal test-time scaling strategy according to task difficulty is crucial, as it can yield greater performance gains than increasing model parameters. Beyond language models, recent research has demonstrated that efficient test-time scaling can also significantly enhance the capabilities of small VLMs [<xref ref-type="bibr" rid="ref37">37</xref>].</p></sec><sec id="s1-2-2"><title>LLMs and VLMs for the Medical Domain</title><p>The medical domain has advanced through specialized models using various training paradigms. UltraMedical [<xref ref-type="bibr" rid="ref27">27</xref>] provides a suite of biomedical LLMs fine-tuned on 410,000 high-quality instructions with preference annotations, achieving state-of-the-art performance through SFT and iterative preference learning. HuatuoGPT-o1 [<xref ref-type="bibr" rid="ref29">29</xref>] uses medical problems with a medical verifier to guide complex reasoning and RL, outperforming baselines using only 40K problems.</p><p>Extending test-time scaling to medicine, m1 [<xref ref-type="bibr" rid="ref34">34</xref>] adapts s1&#x2019;s methodology using small datasets with reasoning traces and thinking token budgets, enabling lightweight models under 10B parameters to achieve state-of-the-art medical reasoning with a 4K token budget.</p><p>To achieve comprehensive medical understanding, multimodal models have recently emerged. HuatuoGPT-Vision [<xref ref-type="bibr" rid="ref38">38</xref>] integrates visual and textual medical knowledge as a 34B multimodal LLM trained on 1.3 million medical VQA (visual question answering) samples. MedGemma [<xref ref-type="bibr" rid="ref39">39</xref>] is Google&#x2019;s open-source medical AI collection combining multimodal capabilities, available in 4B multimodal built on Gemma 3 architecture.</p><p>Beyond architecture, RL has been increasingly leveraged for test-time scaling and improved model robustness in vision-language medical models. MedVLM-R1 [<xref ref-type="bibr" rid="ref33">33</xref>] uses RL to generate natural language reasoning alongside answers. Med-R1 [<xref ref-type="bibr" rid="ref32">32</xref>] uses group relative policy optimization [<xref ref-type="bibr" rid="ref23">23</xref>] to improve generalizability across various medical imaging modalities, achieving a 29.94% accuracy improvement. Both models demonstrate RL effectiveness in medical AI, with MedVLM-R1 focusing on reasoning transparency and Med-R1 emphasizing cross-modality generalization.</p></sec></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Datasets</title><p>In our experiments, we evaluate LLMs on five medical benchmark datasets, including four medical question answering (QA) datasets and one medical calculation dataset. Characteristics of all medical benchmark datasets used in this study are summarized in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Medical benchmark datasets for our experiments.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Name</td><td align="left" valign="top">Type</td><td align="left" valign="top">Description</td><td align="left" valign="top">Answer format</td><td align="left" valign="top">Samples (n)</td><td align="left" valign="top">Difficulty</td></tr></thead><tbody><tr><td align="left" valign="top">Text-only</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>PubMedQA</td><td align="left" valign="top">Medical QA<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top">Research questions with corresponding abstracts</td><td align="left" valign="top">One of yes/no/maybe</td><td align="left" valign="top">500</td><td align="left" valign="top">Easy</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MedQA</td><td align="left" valign="top">Medical QA</td><td align="left" valign="top">Questions based on the USMLE<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">One of five options (A to E)</td><td align="left" valign="top">1273</td><td align="left" valign="top">Intermediate</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MedBullets</td><td align="left" valign="top">Medical QA</td><td align="left" valign="top">Questions based on the USMLE</td><td align="left" valign="top">One of five options (A to E)</td><td align="left" valign="top">308</td><td align="left" valign="top">Intermediate</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MedXpertQA (text)</td><td align="left" valign="top">Medical QA</td><td align="left" valign="top">Expert-level examination questions</td><td align="left" valign="top">One of ten options (A to J)</td><td align="left" valign="top">2450</td><td align="left" valign="top">Difficult</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MedCalc-Bench</td><td align="left" valign="top">Medical calculation</td><td align="left" valign="top">Patient notes and corresponding questions</td><td align="left" valign="top">Decimal, integer, date, time</td><td align="left" valign="top">1047</td><td align="left" valign="top">Difficult</td></tr><tr><td align="left" valign="top">Vision-text</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>OmniMedVQA</td><td align="left" valign="top">Medical VQA<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">Images from various modalities and corresponding questions</td><td align="left" valign="top">One of two/three/four options</td><td align="left" valign="top">5000</td><td align="left" valign="top">Easy</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MedXpertQA (multimodal)</td><td align="left" valign="top">Medical VQA</td><td align="left" valign="top">Expert-level examination questions with corresponding images</td><td align="left" valign="top">One of five options (A to E)</td><td align="left" valign="top">2000</td><td align="left" valign="top">Difficult</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>QA: question answering.</p></fn><fn id="table1fn2"><p><sup>b</sup>USMLE: United States Medical Licensing Examination.</p></fn><fn id="table1fn3"><p><sup>c</sup>VQA: visual question answering.</p></fn></table-wrap-foot></table-wrap><p>PubMedQA [<xref ref-type="bibr" rid="ref40">40</xref>] is a medical QA dataset comprising 500 research questions, each paired with a relevant abstract. The model must answer each question with one of three choices: &#x201C;yes,&#x201D; &#x201C;no,&#x201D; or &#x201C;maybe.&#x201D;</p><p>MedQA [<xref ref-type="bibr" rid="ref41">41</xref>] includes medical QA pairs derived from textbooks. In our study, we use a subset of 1273 multiple-choice questions based on the USMLE (United States Medical Licensing Examination), each with five answer options.</p><p>MedBullets [<xref ref-type="bibr" rid="ref42">42</xref>] is another USMLE-style medical QA dataset, consisting of 308 multiple-choice questions, each with five options, similar in format to MedQA.</p><p>MedXpertQA (text) [<xref ref-type="bibr" rid="ref43">43</xref>] is a recently proposed medical QA benchmark that includes both textual and multimodal (text and image) questions. For our LLM experiments, we use only the text-based subset, which contains 2450 questions with ten answer choices each.</p><p>MedCalc-Bench [<xref ref-type="bibr" rid="ref44">44</xref>] is a medical calculation benchmark comprising 1047 questions, each accompanied by a corresponding patient note. The dataset is designed to evaluate models&#x2019; ability to perform clinical calculations based on contextual patient information.</p><p>We define the difficulty of the medical QA datasets in ascending order as follows: PubMedQA, MedQA, MedBullets, and MedXpertQA, based on previous studies [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref45">45</xref>], the number of choices, and the level of reasoning required to answer the questions. Additionally, given that prior studies have demonstrated the substantial reasoning demands involved in calculations [<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref20">20</xref>], we consider MedCalc-Bench a difficult dataset that requires extensive reasoning.</p><p>Next, we evaluate VLMs on two medical multimodal datasets.</p><p>OmniMedVQA [<xref ref-type="bibr" rid="ref46">46</xref>] is a large-scale benchmark designed for evaluating medical VLMs. OmniMedQVA consists of simple and short questions based on provided images, making its overall difficulty relatively easy. For our experiments, we randomly sample 5000 questions spanning three categories that require relatively more reasoning: disease diagnosis, lesion grading, and other biological attributes.</p><p>Regarding MedXpertQA (multimodal), for our VLM experiments, we use a multimodal subset of the MedXpertQA dataset, which includes 2000 multiple-choice questions, each with five answer options. Each question is accompanied by one to six associated medical images. As the models have to process multiple images simultaneously and comprehend the contextual information in the questions, we consider MedXpertQA a challenging task.</p></sec><sec id="s2-2"><title>Models</title><p>We use the following LLMs for our experiments.</p><p>Regarding general instruction-tuned LLMs, we evaluate instruction-tuned versions of Llama 3 [<xref ref-type="bibr" rid="ref5">5</xref>] (Llama 3-3B, 8B, 70B) and Qwen2.5 [<xref ref-type="bibr" rid="ref8">8</xref>] (Qwen2.5-3B, 7B, 32B, 72B) as representative general-purpose LLMs. Both models have been fine-tuned on instruction-following datasets. Llama 3 is trained using SFT followed by RL from human feedback [<xref ref-type="bibr" rid="ref47">47</xref>], whereas Qwen2.5 is fine-tuned using a combination of SFT, direct preference optimization [<xref ref-type="bibr" rid="ref22">22</xref>], and group relative policy optimization [<xref ref-type="bibr" rid="ref23">23</xref>]. In addition, we evaluate the Qwen3-Instruct models (Qwen3-8B, 30B-IT), which represent the updated nonthinking mode of the Qwen3 model [<xref ref-type="bibr" rid="ref48">48</xref>].</p><p>Regarding general reasoning LLMs, we also evaluate the Qwen3-Thinking models (Qwen3-8B, 30B-TH), which are scaled to enhance the thinking capability of the Qwen3 models, specifically targeting improved quality and depth of reasoning [<xref ref-type="bibr" rid="ref48">48</xref>]. Next, we include distilled versions of DeepSeek-R1 [<xref ref-type="bibr" rid="ref9">9</xref>] (DeepSeek-R1-7B, 8B, 32B, 70B) as representative models optimized for general reasoning capabilities. While the original DeepSeek-R1 is trained using both SFT and RL, the distilled versions are trained solely via SFT using a reasoning dataset generated by the original DeepSeek-R1 model. Finally, we include the gpt-oss models [<xref ref-type="bibr" rid="ref49">49</xref>] (gpt-oss-20B, 120B), OpenAI&#x2019;s open-source series trained via SFT and RL.</p><p>Regarding medical LLMs, we evaluate LLMs specifically trained for the medical domain.</p><p>UltraMedical [<xref ref-type="bibr" rid="ref27">27</xref>] is a medical LLM fine-tuned using SFT followed by preference optimization techniques such as direct preference optimization [<xref ref-type="bibr" rid="ref22">22</xref>] or Kahneman-Tversky optimization [<xref ref-type="bibr" rid="ref50">50</xref>]. We use the 8B and 70B variants of UltraMedical in our experiments (UltraMedical-8B, 70B).</p><p>Regarding medical reasoning LLMs, we also evaluate LLMs specifically designed for medical reasoning tasks. HuatuoGPT-o1 [<xref ref-type="bibr" rid="ref29">29</xref>] (HuatuoGPT-7B, 8B, 70B, 72B) is a medical LLM trained on synthetic medical problems featuring complex chain-of-thought reasoning. The model is trained using SFT and RL via proximal policy optimization [<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>In addition, we include m1 [<xref ref-type="bibr" rid="ref34">34</xref>], a medical reasoning model trained with curated chain-of-thought&#x2013;style data using SFT, but without RL. In our experiments, we use versions of m1 that are trained on a dataset of 1000 samples, which are denoted as m1-7B, 32B in this paper.</p><p>Regarding general VLMs, we evaluate five general-purpose VLMs: Llama 3-Vision [<xref ref-type="bibr" rid="ref5">5</xref>] (Llama 3-Vision-11B, 90B), Qwen2.5-VL [<xref ref-type="bibr" rid="ref51">51</xref>] (Qwen2.5-VL-7B, 32B), Qwen3-VL [<xref ref-type="bibr" rid="ref52">52</xref>] (Qwen3-VL-8B, 30B-IT), Gemma 3 [<xref ref-type="bibr" rid="ref53">53</xref>] (Gemma 3-12B, 27B), and LLaVA [<xref ref-type="bibr" rid="ref13">13</xref>] (Large Language and Vision Assistant; LLaVA-7B, 13B). For Llama, Qwen, and Gemma models, we use instruction-tuned versions.</p><p>Regarding general reasoning VLMs, we also include general-domain models explicitly trained for reasoning tasks. First, the thinking version of Qwen3-VL [<xref ref-type="bibr" rid="ref52">52</xref>] (Qwen3-VL-8B, 30B-TH) is evaluated. Next, LLaVA-CoT [<xref ref-type="bibr" rid="ref54">54</xref>] (Large Language and Vision Assistant&#x2013;Chain-of-Thought; LLaVA-CoT-11B) is a VLM fine-tuned with synthetic chain-of-thought data using SFT. Finally, we evaluate the preview version of QVQ [<xref ref-type="bibr" rid="ref55">55</xref>] (Qwen With Vision and Questions; QVQ-72B), a large-scale multimodal reasoning model built upon Qwen2-VL-72B [<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>Regarding medical VLMs, we evaluate VLMs specifically tuned for the medical domain. MedGemma [<xref ref-type="bibr" rid="ref39">39</xref>] (MedGemma-4B, 27B) is a medical variant of Gemma 3 trained on medical text and images. In our experiment, we use the instruction-tuned version of MedGemma. HuatuoGPT-Vision [<xref ref-type="bibr" rid="ref38">38</xref>] (HuatuoGPT-Vision-7B, 34B) is another medical VLM fine-tuned on medical VQA data.</p><p>Regarding medical reasoning VLMs, lastly, we include the medical reasoning VLM, QoQ-Med [<xref ref-type="bibr" rid="ref31">31</xref>], in our evaluation. QoQ-Med is trained on the CLIMB (clinical large-scale integrative multimodal benchmark) [<xref ref-type="bibr" rid="ref56">56</xref>], a comprehensive dataset that incorporates diverse types of medical data, including ECG (electrocardiogram; 1D), chest X-rays (2D), and magnetic resonance imaging scans (3D). To enhance multimodal reasoning capabilities, the authors of a study [<xref ref-type="bibr" rid="ref31">31</xref>] proposed a domain-aware group relative policy optimization, which applies hierarchical scaling strategies based on the domain of the input data. In our experiments, we evaluate two versions of QoQ-Med: QoQ-Med-7B and QoQ-Med-32B.</p><p><xref ref-type="table" rid="table2">Table 2</xref> summarizes the models used in our experiments and their key characteristics. In our experiments, we use a 4-bit quantized version of the models when the number of model parameters is 70B or more.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>LLMs<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> and VLMs<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> for our experiments.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Model name</td><td align="left" valign="top">Model type</td><td align="left" valign="top">Domain</td><td align="left" valign="top">Base model</td><td align="left" valign="top">Dataset</td><td align="left" valign="top">Training method</td><td align="left" valign="top">Model size</td></tr></thead><tbody><tr><td align="left" valign="top">LLM</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Llama 3-Instruct</td><td align="left" valign="top">General</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">Llama 3</td><td align="left" valign="top">Instruction datasets</td><td align="left" valign="top">SFT<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>+RLHF<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">3B, 8B, 70B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Qwen2.5-Instruct</td><td align="left" valign="top">General</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">Qwen2.5</td><td align="left" valign="top">Instruction datasets</td><td align="left" valign="top">SFT+DPO<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup>+GRPO<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="left" valign="top">3B, 7B, 32B, 72B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Qwen3-Instruct</td><td align="left" valign="top">General</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">Qwen3</td><td align="left" valign="top">Unknown</td><td align="left" valign="top">Unknown</td><td align="left" valign="top">4B, 30B (3B activated)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Qwen3-Thinking</td><td align="left" valign="top">General</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">Qwen3</td><td align="left" valign="top">Unknown</td><td align="left" valign="top">Unknown</td><td align="left" valign="top">4B, 30B (3B activated)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-R1-Distill</td><td align="left" valign="top">General</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">Llama 3 or Qwen2.5</td><td align="left" valign="top">Reasoning data generated by DeepSeek-R1</td><td align="left" valign="top">SFT</td><td align="left" valign="top">7B, 8B, 32B, 70B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>gpt-oss</td><td align="left" valign="top">General</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">N/A<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup></td><td align="left" valign="top">Text-only dataset with a focus on STEM<sup><xref ref-type="table-fn" rid="table2fn8">h</xref></sup>, coding, and general knowledge</td><td align="left" valign="top">SFT+CoT<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup> RL<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">20B, 120B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>UltraMedical</td><td align="left" valign="top">Medical</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">Llama 3</td><td align="left" valign="top">Synthetic (by GPT-4 [<xref ref-type="bibr" rid="ref2">2</xref>]) and manually curated medical data</td><td align="left" valign="top">SFT+DPO or KTO<sup><xref ref-type="table-fn" rid="table2fn11">k</xref></sup></td><td align="left" valign="top">8B, 70B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>HuatuoGPT-o1</td><td align="left" valign="top">Medical</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">Llama 3 or Qwen2.5</td><td align="left" valign="top">Medical data with synthetic CoT reasoning (by GPT-4o [<xref ref-type="bibr" rid="ref6">6</xref>])</td><td align="left" valign="top">SFT+PPO<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">7B, 8B, 70B, 72B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>m1</td><td align="left" valign="top">Medical</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">Qwen2.5</td><td align="left" valign="top">Curated medical data with synthetic reasoning (by DeepSeek-R1)</td><td align="left" valign="top">SFT</td><td align="left" valign="top">7B, 32B</td></tr><tr><td align="left" valign="top">VLM</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Llama 3-Vision-Instruct</td><td align="left" valign="top">General</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">Llama 3-Vision</td><td align="left" valign="top">Vision-language instruction dataset</td><td align="left" valign="top">SFT+RLHF</td><td align="left" valign="top">11B, 90B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Qwen2.5-VL-Instruct</td><td align="left" valign="top">General</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">Qwen2.5-VL</td><td align="left" valign="top">Vision-language instruction dataset</td><td align="left" valign="top">SFT+DPO</td><td align="left" valign="top">7B, 32B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemma 3-it</td><td align="left" valign="top">General</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">Gemma 3</td><td align="left" valign="top">Vision-language instruction dataset</td><td align="left" valign="top">SFT+RLHF</td><td align="left" valign="top">12B, 27B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLaVA<sup><xref ref-type="table-fn" rid="table2fn13">m</xref></sup></td><td align="left" valign="top">General</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">Llama 2 or Vicuna [<xref ref-type="bibr" rid="ref3">3</xref>]</td><td align="left" valign="top">Vision-language instruction dataset</td><td align="left" valign="top">SFT</td><td align="left" valign="top">7B, 13B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Qwen3-VL-Instruct</td><td align="left" valign="top">General</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Vision-language instruction dataset+reasoning data</td><td align="left" valign="top">SFT+Distillation+SAPO<sup><xref ref-type="table-fn" rid="table2fn14">n</xref></sup> [<xref ref-type="bibr" rid="ref57">57</xref>]</td><td align="left" valign="top">8B, 30B (3B activated)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Qwen3-VL-Thinking</td><td align="left" valign="top">General</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Vision-language instruction dataset+reasoning data</td><td align="left" valign="top">SFT+Distillation+SAPO [<xref ref-type="bibr" rid="ref57">57</xref>]</td><td align="left" valign="top">8B, 30B (3B activated)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLaVA-CoT<sup><xref ref-type="table-fn" rid="table2fn15">o</xref></sup></td><td align="left" valign="top">General</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">Llama 3-Vision-Instruct</td><td align="left" valign="top">Reasoning data generated by GPT-4o</td><td align="left" valign="top">SFT</td><td align="left" valign="top">11B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>QVQ<sup><xref ref-type="table-fn" rid="table2fn16">p</xref></sup>-Preview</td><td align="left" valign="top">General</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">Qwen2-VL</td><td align="left" valign="top">Unknown</td><td align="left" valign="top">Unknown</td><td align="left" valign="top">72B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MedGemma-it</td><td align="left" valign="top">Medical</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">MedGemma</td><td align="left" valign="top">Medical image and text data</td><td align="left" valign="top">Unknown</td><td align="left" valign="top">4B, 27B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>HuatuoGPT-Vision</td><td align="left" valign="top">Medical</td><td align="left" valign="top">Nonreasoning</td><td align="left" valign="top">LLaVA or Yi [<xref ref-type="bibr" rid="ref58">58</xref>]</td><td align="left" valign="top">Medical VQA<sup><xref ref-type="table-fn" rid="table2fn17">q</xref></sup> dataset</td><td align="left" valign="top">SFT</td><td align="left" valign="top">7B, 34B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>QoQ-Med</td><td align="left" valign="top">Medical</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">Qwen2.5-VL</td><td align="left" valign="top">Multimodal (1D, 2D, 3D) medical dataset [<xref ref-type="bibr" rid="ref56">56</xref>]</td><td align="left" valign="top">DRPO<sup><xref ref-type="table-fn" rid="table2fn18">r</xref></sup></td><td align="left" valign="top">7B, 32B</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table2fn2"><p><sup>b</sup>VLM: vision-language model.</p></fn><fn id="table2fn3"><p><sup>c</sup>SFT: supervised fine-tuning. </p></fn><fn id="table2fn4"><p><sup>d</sup>RLHF: reinforcement learning from human feedback.</p></fn><fn id="table2fn5"><p><sup>e</sup>DPO: direct preference optimization.</p></fn><fn id="table2fn6"><p><sup>f</sup>GRPO: group relative policy optimization.</p></fn><fn id="table2fn7"><p><sup>g</sup>N/A: not applicable.</p></fn><fn id="table2fn8"><p><sup>h</sup>STEM: Science, Technology, Engineering, and Mathematics. </p></fn><fn id="table2fn9"><p><sup>i</sup>CoT: chain-of-thought.</p></fn><fn id="table2fn10"><p><sup>j</sup>RL: reinforcement learning.</p></fn><fn id="table2fn11"><p><sup>k</sup>KTO: Kahneman-Tversky optimization.</p></fn><fn id="table2fn12"><p><sup>l</sup>PPO: proximal policy optimization.</p></fn><fn id="table2fn13"><p><sup>m</sup>LLaVA: Large Language and Vision Assistant.</p></fn><fn id="table2fn14"><p><sup>n</sup>SAPO: soft adaptive policy optimization.</p></fn><fn id="table2fn15"><p><sup>o</sup>LLaVA-CoT: Large Language and Vision Assistant&#x2013;Chain-of-Thought. </p></fn><fn id="table2fn16"><p><sup>p</sup>QVQ: Qwen With Vision and Questions. </p></fn><fn id="table2fn17"><p><sup>q</sup>VQA: visual question answering.</p></fn><fn id="table2fn18"><p><sup>r</sup>DRPO: domain-aware group relative policy optimization.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-3"><title>Experimental Setting</title><sec id="s2-3-1"><title>Prompts for Models</title><p>In our experiments, we use task-specific prompts tailored to each dataset (Figures S1 and S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Consistent with the approach in another study [<xref ref-type="bibr" rid="ref34">34</xref>], we append a common instruction at the end of each prompt, &#x201C;return your final response within \boxed{{}}.,&#x201D; to extract the final answer of the model clearly. Additionally, for models not explicitly designed for reasoning (eg, Llama 3 and Qwen2.5), we explicitly include the phrase &#x201C;let&#x2019;s think step by step&#x201D; at the end of the prompt to encourage step-by-step reasoning by chain-of-thought during inference. An ablation study demonstrating the critical role of reasoning triggers in nonreasoning models is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> as Figure S7.</p></sec><sec id="s2-3-2"><title>Test-Time Scaling Across Different Token Budgets</title><p>We investigate test-time scaling in the medical domain by varying the token budget, specifically by adjusting the maximum sequence length allowed for model generation (<xref ref-type="fig" rid="figure2">Figure 2B</xref>). If the model reaches this maximum sequence length, its output is truncated accordingly. In cases where the model does not produce a final answer within the allotted length, we append &#x201C;\boxed{{&#x201D; to the end of the response and force the model to generate the final answer, which is then used for evaluation. Conversely, if the model completes its reasoning and provides a final response before reaching the token limit, we do not force it to continue generating tokens. To assess how efficiently models use the available token budget, we compute the average number of tokens used during the reasoning process and compare this across models and datasets. We set the temperature to 0 for the token budget experiments.</p><p>Next, we evaluate model accuracy across different token budgets. For the medical QA benchmarks, accuracy is measured by the number of final answers that exactly match the ground truth. In contrast, for the MedCalc-Bench dataset, we apply task-specific evaluation criteria: for equation-based calculation problems with decimal answers, a response is considered correct if it falls within a 5% error margin of the correct value; for all other problem types, only exact matches with the correct answer are counted as correct.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Overview of test-time scaling strategies: (A) no scaling, (B) increasing the token budget, (C) iterative sequential scaling, (D) parallel scaling, and (E) hybrid sequential-parallel scaling.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90693_fig02.png"/></fig></sec><sec id="s2-3-3"><title>Sequential and Parallel Scaling</title><p>Several studies have compared sequential scaling and parallel scaling in the general domain [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref20">20</xref>], with mixed findings. Motivated by this, we investigate which approach is more effective in the medical domain.</p><p>For iterative sequential scaling, responses are not generated all at once within the full token budget. Instead, the model is initially prompted to generate a response within a limited budget (512 tokens). Subsequently, the model is iteratively prompted to revise and extend its previous response. To initiate this revision process, we append the token &#x201C;wait&#x201D; at the end of each response, prompting the model to continue and refine its reasoning in subsequent steps, as proposed in previous works [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref34">34</xref>].</p><p>In contrast, parallel scaling involves generating multiple responses simultaneously. To determine the final answer from these responses, we apply the shortest majority vote strategy [<xref ref-type="bibr" rid="ref20">20</xref>]. This strategy computes a specific score for each unique answer category. The score is calculated by dividing the number of solutions in that category by the logarithm of their average length. The final answer is chosen from the category with the highest score. This mathematical formulation addressed a known characteristic of certain reasoning models where performance deteriorates with overly prolonged solution lengths. By penalizing excessive length logarithmically while heavily weighting answer frequency, this approach effectively balances robust consensus with an empirical safeguard against hallucination or error accumulation. To explore the optimal combination of test-time scaling strategies, we additionally experiment with a hybrid approach that integrates both sequential and parallel scaling. This allows us to investigate whether combining iterative refinement with diversity from parallel sampling can lead to improved performance in medical reasoning tasks.</p><p><xref ref-type="fig" rid="figure2">Figure 2A, 2C, and 2D</xref> illustrate the concepts of no test-time scaling, sequential and parallel scaling, respectively. To determine whether a model has chosen the correct reasoning path, we extract the conclusion from the generated output and strictly compare it against the ground truth label of the dataset. We evaluate model performance using two metrics: accuracy and coverage. Accuracy measures the model&#x2019;s success in selecting the correct reasoning path and arriving at the correct conclusion, thereby reflecting its decision efficiency. Coverage, on the other hand, measures the proportion of correct answers found in all intermediate outputs, whether across sequential steps or parallel samples, and captures the model&#x2019;s capacity to explore a diverse reasoning space.</p><p>A model with high coverage but low final accuracy suggests it can generate valid reasoning paths but often fails to select them as the final answer. This implies strong exploratory capacity but a tendency to diverge from correct conclusions due to overthinking, inconsistent logic, or ineffective selection strategies.</p><p>To control generation variability, we set the temperature to 0 for sequential scaling and 0.7 for parallel scaling.</p></sec><sec id="s2-3-4"><title>User-Driven Factor</title><p>We also investigate whether test-time scaling enhances the robustness of LLMs to user-driven factors using three medical QA benchmark datasets: MedQA, Medbullets, and MedXpertQA. Building on prior work [<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref60">60</xref>], we focus on scenarios where the user injects misleading information into the prompt.</p><p>In these experiments, we insert misleading hints into the prompt, presented as another physician&#x2019;s opinion favoring an incorrect answer. To maximize confusion, we use GPT-4o to select the most semantically plausible incorrect option. We then incorporate this misleading hint into the prompt, varying two factors: the tone of the comment (hedged vs definitive) and the expertise level of the physician (novice vs expert). The following examples illustrate the different conditions: (1) hedged tone, novice physician: &#x201C;Comment from another novice physician: I think the answer is probably A.&#x201D; (2) Hedged tone, expert physician: &#x201C;Comment from another expert physician: I think the answer is probably A.&#x201D; (3) Definitive tone, novice physician: &#x201C;Comment from another novice physician: I am confident the answer is A.&#x201D; (4) Definitive tone, expert physician: &#x201C;Comment from another expert physician: I am confident the answer is A.&#x201D;</p></sec></sec><sec id="s2-4"><title>Ethical Considerations</title><p>This study does not involve human participants, animal participants, medical records, or any personally identifiable information. All experiments were conducted exclusively using publicly available and anonymized medical benchmark datasets to evaluate LLMs. Therefore, this study does not constitute human participants research and did not require formal review or approval by an institutional review board or a local ethics committee, in accordance with standard institutional and national policies regarding the secondary analysis of publicly available data.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Test-Time Scaling Across Different Token Budgets</title><sec id="s3-1-1"><title>Experiments With LLMs</title><p>First, we investigate whether test-time scaling of LLMs is effective in the medical domain by increasing the token budget. <xref ref-type="fig" rid="figure3">Figure 3</xref> presents the average number of reasoning tokens used under different token budgets, where the first row displays Llama-based LLMs and the second row shows Qwen-based LLMs. <xref ref-type="fig" rid="figure4">Figure 4</xref> presents the accuracy of various models across different token budgets.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Test-time scaling of various large language models across different token budgets: average number of reasoning tokens used. * in the legend indicates models that are 4-bit quantized.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90693_fig03.png"/></fig><p>Regarding nonreasoning models, as illustrated in <xref ref-type="fig" rid="figure3">Figure 3</xref>, LLMs that lack explicit reasoning capabilities typically use only a small portion of the available token budget (approximately 500 to 1000 tokens), even when a larger budget is provided. Consequently, as shown in <xref ref-type="fig" rid="figure4">Figure 4</xref>, their accuracy quickly saturates and does not improve with increased token budgets. This effect is especially evident in smaller models with fewer than 10B parameters. These observations suggest that simply increasing the token budget at test time is unlikely to enhance performance for intrinsic nonreasoning models.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Test-time scaling of various LLMs (large language models) across different token budgets: accuracy of LLMs as a function of token budget. * in the legend indicates models that are 4-bit quantized.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90693_fig04.png"/></fig><p>For reasoning models, most show minimal or no improvement in performance on the easier task (PubMedQA) as the token budget increases. However, as the task difficulty increases (defined in the Methods section), reasoning models tend to use more reasoning tokens. Notably, the degree and manner of this token usage vary across different reasoning models, each exhibiting distinct trends in response to increasing task complexity.</p><p>For example, DeepSeek-R1 or Qwen3-TH models tend to use more tokens for reasoning as the token budget increases, and perform better with increased token budgets. This effect is more pronounced in models with larger parameter counts. However, the smaller variants of DeepSeek-R1 (7B or 8B) perform worse than other general-domain models of similar size. We hypothesize that this is because the versions of DeepSeek-R1 used in our experiments are distilled models trained on synthetic data (generated by the original DeepSeek-R1) using SFT, rather than through RL. As a result, these smaller distilled models may fail to fully retain the reasoning capabilities of the original model, leading to their lower performance.</p><p>The gpt-oss models demonstrate adaptive token usage based on task complexity. On simpler benchmarks such as PubMedQA and MedQA, these models maintain relatively low token usage even when provided with large budgets, suggesting that extensive reasoning is not triggered for easier queries. However, as task difficulty increases, the models exhibit an upward trend in token consumption, using a large portion of the available budget. Crucially, this increased usage of reasoning tokens correlates with performance gains. The gpt-oss models achieve higher accuracy on challenging tasks as the token budget expands. To further illustrate this direct relationship, we provide an explicit plot of actual token consumption against accuracy for all evaluated models in Figure S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. This visualization confirms that for reasoning-capable models, increased token usage is strongly correlated with improved accuracy, particularly on complex tasks.</p><p>HuatuoGPT-o1 generally exhibits strong performance across all experiments. However, despite being categorized as a reasoning model, it tends to use a relatively small number of tokens during its reasoning process. As a result, it demonstrates limited performance gains from test-time scaling via increased token budgets compared to other reasoning models.</p><p>In contrast, m1 effectively uses a larger portion of the available token budget and exhibits substantial performance improvements as the token budget increases. Notably, on the most challenging QA dataset in our study, MedXpertQA, m1-32B demonstrates clear test-time scalability, outperforming other models with even larger parameter counts as the token budget increases. We attribute the difference between HuatuoGPT-o1 and m1 to differences in the nature of their respective training data. While the reasoning traces used to train HuatuoGPT-o1 were generated by GPT-4o [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref29">29</xref>], those used for m1 were synthesized by DeepSeek-R1 [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. These underlying models differ in their reasoning styles and output structures, likely leading to variations in the length and structure of chain-of-thought traces and influencing how each model uses tokens during inference.</p><p>Regarding general vs medical models, as expected, medical LLMs generally outperform general domain LLMs on medical QA tasks, except for the recently released Qwen3 variants and gpt-oss. However, in the case of MedCalc-Bench, the accuracy of medical models is generally lower than that of general domain models. This discrepancy suggests that MedCalc-Bench primarily rewards general numerical or procedural reasoning applied within a medical context, rather than the qualitative clinical reasoning that centralizes specialized medical training. Therefore, the stronger performance of general-purpose models does not necessarily imply deficient medical reasoning overall. Instead, it indicates that the benchmark captures a narrower procedural capability that is less tied to domain-specific clinical expertise. Nevertheless, performance can still be improved through test-time scaling with an increased token budget.</p><p>In summary, test-time scaling through increased token budgets can improve performance for certain reasoning LLMs. However, many existing LLMs that benefit from this approach are not specifically designed for the medical domain. This highlights the need to explore alternative test-time scaling strategies better suited for medical applications.</p></sec><sec id="s3-1-2"><title>Experiments With VLMs</title><p>Next, we investigate the test-time scaling in the medical domain using VLMs. Similar to the previous experiment, we analyze the impact of increasing token budgets on model performance.</p><p><xref ref-type="fig" rid="figure5">Figure 5A</xref> shows the average number of tokens used by VLMs during reasoning. Consistent with our findings from LLMs, most models do not significantly increase their token usage, even when given a higher token budget. In contrast, models such as the Qwen3-VL variants, MedGemma-27B, and QVQ exhibit a distinct upward trend in token usage as the available token budget increases, indicating their capacity to leverage additional computational resources for extended reasoning.</p><p><xref ref-type="fig" rid="figure5">Figure 5B</xref> presents model accuracy on two datasets across different token budgets. On OmniMedVQA, which is relatively simple and less demanding in reasoning, most models do not show any performance improvement as the token budget increases. In contrast, on MedXpertQA, a dataset requiring more complex reasoning, we observe a more noticeable, though still limited, test-time scaling effect. Conversely, models that exhibit an upward trend in token usage demonstrate clear performance gains from increased token budgets on MedXpertQA, a correlation that is particularly pronounced in large-scale models with 27B or more parameters.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Test-time scaling of VLMs (vision-language model) across medical VQA (visual question answering) benchmark: (A) average number of reasoning tokens used and (B) accuracy of LLMs (large language model) as a function of token budget. * in the legend indicates models that are 4-bit quantized.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90693_fig05.png"/></fig><p>Interestingly, while QVQ displays test-time scaling on both benchmarks, its performance is paradoxically lower on the simpler OmniMedVQA and higher on the more challenging MedXpertQA. Upon closer inspection of QVQ&#x2019;s responses, we attribute this to a combination of model limitations and dataset characteristics. Specifically, QVQ frequently fails to process the input images in OmniMedVQA correctly. It sometimes outputs messages such as &#x201C;I can&#x2019;t see the image&#x201D; (Figure S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) or misinterprets prompts as part of the image content (eg, &#x201C;There are also some text elements overlaid on the image, such as &#x2018;You are a helpful assistant&#x2019;&#x201D;; Figure S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Additionally, due to these misinterpretations, QVQ often fails to choose from the provided answer options, instead responding with &#x201C;none of the above&#x201D; (Figure S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), further reducing its accuracy. In contrast, the performance of QVQ on MedXpertQA benefits from the presence of more informative textual clues within the questions, while the questions of OmniMedVQA are very simple (Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). These clues often enable the model to infer the correct answer even without fully using the image, thus reducing the impact of its visual processing limitations. As a result, QVQ achieves higher accuracy despite the greater task complexity.</p><p>Overall, our findings from the VLM experiments are consistent with those from the LLM experiments: test-time scaling, by simply increasing the token budget, provides limited benefits for most VLMs. This approach shows effectiveness only for some models, and even then, the performance gains in VLMs are considerably smaller than those observed in LLMs. Notably, as in the case of LLMs, the effectiveness of test-time scaling in VLMs is more pronounced on more difficult tasks, while it has little to no impact on simpler tasks.</p><p>These results can be interpreted from two perspectives. First, unlike LLMs, most of the current VLMs may lack sufficiently developed reasoning capabilities, particularly in leveraging visual cues for complex decision-making. While recent studies have proposed VLMs trained for reasoning, these models are often not explicitly trained to reason effectively using visual information and thus struggle to interpret and analyze complex medical images as effectively as LLMs do with text. However, our findings point to potential pathways for enhancing VLM reasoning capabilities. First, the adoption of advanced posttraining methodologies can significantly reinforce reasoning performance. For instance, the Qwen3-VL models were posttrained using SFT, distillation, and Soft Adaptive Policy Optimization [<xref ref-type="bibr" rid="ref57">57</xref>], an RL method specifically proposed for Qwen3-VL. Consequently, these models exhibit a distinct test-time scaling effect on the challenging MedXpertQS benchmark as the token budget increases. Second, scaling model parameters in conjunction with medical domain training proves beneficial. In our experiment, MedGemma-27B demonstrates improved performance on MedXpertQA as the token budget expands, whereas the smaller MedGemma-4B does not.</p><p>Next, the limitations may stem from the current vision-language medical benchmarks, which may not be well-suited for evaluating reasoning ability. For instance, OmniMedVQA primarily consists of relatively simple image-based questions that require minimal reasoning, thus failing to challenge the model&#x2019;s reasoning capabilities. Conversely, MedXpertQA presents highly challenging questions that often require the simultaneous analysis of multiple medical images. Most contemporary VLMs still struggle with or entirely lack the ability to jointly and effectively reason over such complex visual cues in conjunction with textual information, which may explain the relatively limited performance on this benchmark compared to that of LLMs. Future work should aim to develop stronger medical reasoning VLMs and more rigorous benchmarks for their evaluation.</p></sec></sec><sec id="s3-2"><title>Sequential and Parallel Scaling</title><p>In previous experiments, we observed that test-time scaling, by simply increasing the token budget, is ineffective for many models. Except for certain reasoning models, most do not use the full available token budget during inference. To address this limitation, we investigate alternative scaling strategies, iterative sequential scaling and parallel scaling, as described in the Methods section. In this experiment, we evaluate two medical reasoning LLMs: HuatuoGPT-o1-7B (denoted as HuatuoGPT-7B) and m1-7B-1k (denoted as m1-7B).</p><p><xref ref-type="fig" rid="figure6">Figure 6</xref> presents a comparison between sequential and parallel test-time scaling across multiple medical benchmarks. On PubMedQA, the easiest QA task, both models show a decline in accuracy as the number of sequential sampling steps increases. Moreover, the accuracy achieved with parallel scaling is higher than that obtained by simply increasing the token budget. This suggests that additional iterations may not be beneficial for simpler tasks; in fact, excessive revisions can lead the models to deviate from initially correct answers. In particular, for m1, accuracy decreases even though coverage increases with more sequential steps, indicating that while the model may identify the correct answer during sequential scaling, unnecessary revisions can lead to an incorrect final output.</p><p>For MedQA and MedBullets, HuatuoGPT-o1 maintains the same trend, showing limited improvement from either sequential scaling or increased token budgets. In contrast, m1 achieves its highest accuracy when the token budget is evenly divided between sequential and parallel sampling (1:1 ratio), with performance comparable to that of simple token budget expansion. These findings suggest that for tasks of moderate difficulty, sequential scaling may offer little advantage for models that naturally generate shorter outputs. For these models, the performance gains from parallel scaling on simpler tasks are driven primarily by reasoning path diversity and answer consistency through majority voting, rather than the brevity of the reasoning itself. In comparison, models such as m1, which rely on deeper reasoning, benefit more from a balanced approach that combines diverse generation with iterative refinement.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Comparison of sequential and parallel test-time scaling of LLMs (large language models) across medical benchmarks. Dotted lines indicate accuracy with an increased token budget to (8192). Shaded regions represent 95% CIs.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90693_fig06.png"/></fig><p>On the more challenging MedXpertQA benchmark, which demands complex reasoning, HuatuoGPT-o1 shows relatively consistent performance across all scaling strategies, including increased token budget, sequential scaling, and parallel scaling. In contrast, m1 achieves its highest accuracy and coverage with fully sequential scaling, slightly surpassing its performance under the increased token budget condition.</p><p>On MedCalc-Bench, which also involves complex reasoning but with distinct characteristics, both models show improved performance as the number of sequential samples increases. However, the highest accuracy is attained when applying test-time scaling via a larger token budget. These findings suggest that step-by-step sequential refinement benefits reasoning-intensive question-answering tasks, while generating a single, extended reasoning path is more effective for tasks requiring precise calculations.</p><p>In summary, under constrained token budgets, parallel scaling proves highly effective for simpler tasks. This advantage stems from the model exploring a diverse set of reasoning paths and arriving at a reliable conclusion through answer consistency. In contrast, for models that rely on more extensive reasoning, sequential scaling or increasing the token budget becomes increasingly advantageous as task complexity grows.</p><p>In addition to evaluating task performance, we analyzed the computational cost, specifically the inference latency (wall-clock time), associated with each scaling strategy to assess their viability for clinical deployment. As illustrated in the bottom row of <xref ref-type="fig" rid="figure6">Figure 6</xref>, increasing the proportion of sequential reasoning results in a substantial increase in processing time per sample. This delay occurs because sequential scaling relies on iterative generation, where each step must wait for the previous one to complete, while the context length continuously expands. Conversely, heavily parallelized configurations maintain significantly lower latency. By using batched inference, parallel scaling processes multiple samples simultaneously without a proportional increase in wall-clock time. This quantitative analysis reveals a critical cost-performance trade-off, indicating that parallel scaling is highly efficient for time-sensitive clinical applications.</p></sec><sec id="s3-3"><title>Impact of Test-Time Scaling on User-Driven Factors</title><p>Finally, we investigate the impact of test-time scaling on robustness to user-driven factors. Again, we evaluate two medical LLMs, HuatuoGPT-o1-7B and m1-7B-1k, across two types of test-time scaling strategies. The first strategy simply increases the token budget by a factor of 16 from the default setting, as described in the first experiment. The second strategy combines iterative sequential scaling and parallel scaling using the optimal configuration identified in the second experiment. <xref ref-type="fig" rid="figure7">Figure 7</xref> shows the accuracy of both models under various user-driven perturbations and test-time scaling strategies.</p><p>First, <xref ref-type="fig" rid="figure7">Figure 7A</xref> shows that user-driven perturbation generally degrades the performance of HuatuoGPT-o1 when no test-time scaling strategy is applied. When test-time scaling is applied by simply increasing the token budget, the improvement is minimal, and in some cases, performance even declines. In contrast, applying the optimal test-time scaling strategy significantly improves robustness against user-driven factors in most scenarios. The benefit is particularly pronounced when the task is easier, such as in MedQA. This is likely because HuatuoGPT-o1 tends to generate relatively short responses by default, and merely increasing the token budget does not effectively encourage the model to use the additional tokens for deeper reasoning.</p><p>On the other hand, m1 also exhibits degraded performance in the presence of user-driven factors, as illustrated in <xref ref-type="fig" rid="figure7">Figure 7B</xref>. However, unlike HuatuoGPT-o1, increasing the token budget proves more effective than applying the optimal sequential-parallel scaling strategy for the m1 model. This is due to m1&#x2019;s inherent tendency to generate long chain-of-thought reasoning when solving problems. As sequential-parallel scaling produces shorter responses at each iteration, it may hinder m1 from fully developing its reasoning and reaching a well-considered conclusion.</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Test-time scaling under different user-driven factor types on medical QA (question answering) benchmarks. (A) HuatuoGPT-o1-7B and (B) m1-7B-1k. Error bars represent 95% CIs.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90693_fig07.png"/></fig><p>In summary, model behavior varies depending on the chosen test-time scaling strategy when user-driven factors exist. This highlights the importance of selecting an appropriate test-time scaling method tailored to each model type to enhance robustness against such perturbations. Nonetheless, some common patterns emerge across models. For relatively easy tasks (eg, MedQA), both models can recover performance comparable to that without user-driven interference through test-time scaling. In contrast, for more challenging tasks (eg, MedXpertQA), the performance gap between conditions with and without user-driven factors remains substantial, even when test-time scaling is applied. Notably, models are particularly sensitive to the expertise level of the physician providing the additional input. This suggests that as task complexity increases, LLMs may struggle to initiate reasoning with high confidence, leading them to heavily rely on the initial context or cues provided by the user. Consequently, when such cues are misleading, the model is more likely to follow an incorrect reasoning path and, due to the increased complexity, has greater difficulty recovering and arriving at the correct answer. These findings emphasize the amplified impact of user-driven perturbations in difficult scenarios and highlight the importance of minimizing such perturbations by providing only neutral and essential information when prompting LLMs for complex or high-stakes medical QA tasks.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><p>To contextualize our findings, the primary aim of this study was to establish a domain-specific framework for scaling inference compute in medical AI. We hypothesized that the optimal scaling strategy depends heavily on the reasoning capacity of the model and the complexity of the clinical task. Our overall results broadly confirm these hypotheses. We demonstrated that simply increasing the computational budget during inference is not a universal remedy. Instead, the efficacy of test-time scaling is highly conditional, varying significantly across model architectures, task difficulties, and clinical contexts.</p><p>First, we found that simply increasing the token budget is not a universal solution for enhancing medical LLM or VLM performance. A distinct contrast exists between nonreasoning and reasoning models. Nonreasoning models (eg, Llama 3 and Qwen2.5) exhibited performance saturation, using fewer than 1000 tokens even when provided with larger budgets, resulting in negligible accuracy gains. In contrast, reasoning-optimized models demonstrated adaptive token usage. Notably, the gpt-oss models exhibited behavior aligned with &#x201C;thinking-optimal&#x201D; scaling. They maintained efficient, low token usage for simpler tasks such as PubMedQA but increased consumption for complex benchmarks such as MedXpertQA, which correlated directly with accuracy improvements. Furthermore, the efficiency of this scaling appears to be heavily influenced by the nature of the model&#x2019;s training data. The m1 model, trained on DeepSeek-R1 traces, used more tokens for deep reasoning than HuatuoGPT-o1, which relied on GPT-4o traces.</p><p>Second, test-time scaling for VLMs currently lags behind that of LLMs. Most VLMs failed to leverage increased budgets effectively and showed limited improvements in reasoning depth. Exceptions to this trend were observed in Qwen3-VL variants and MedGemma-27B. This suggests that advanced posttraining methods such as Soft Adaptive Policy Optimization and larger parameter sizes are critical enablers for multimodal test-time scaling. Specifically, we observed that scaling model parameters in conjunction with medical domain training proves beneficial.</p><p>Third, we demonstrated that the optimal scaling strategy is highly task-dependent. For simpler tasks, excessively long reasoning paths are not necessarily beneficial. Iterative sequential scaling proved harmful on easier datasets such as PubMedQA. This performance drop is likely due to compounding errors during unnecessarily extended reasoning. In this case, parallel scaling using majority voting yielded superior results. This success arises from exploring a diverse set of reasoning paths and aggregating them through majority voting to ensure answer consistency. Conversely, reasoning-intensive tasks such as MedXpertQA and calculation tasks such as MedCalc-Bench benefited from extended sequential reasoning.</p><p>Finally, while test-time scaling enhances robustness against use-driven perturbations, it requires model-specific tailoring. We observed that scaling strategies generally mitigated the negative impact of misleading expert hints but often failed to fully restore baseline performance on difficult tasks. Crucially, the effective strategy varied by model architecture. HuatuoGPT-o1 benefited most from a hybrid sequential-parallel approach. However, m1 relies on generating long, uninterrupted reasoning chains and performs best with a simple expansion of the token budget. This underscores the necessity of aligning scaling strategies with both the difficulty of the clinical task and the intrinsic reasoning behaviors of the specific model being deployed.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>Prior studies on test-time scaling in general domains, such as mathematics and coding, have established that increasing inference compute can improve reasoning, though the relationship is not strictly monotonic. These works demonstrate that optimal reasoning lengths exist and that excessive self-revision can sometimes degrade performance [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>]. While these foundational works highlight the need for task-dependent scaling strategies, our study reveals that in the high-stakes medical domain, the limitations of test-time scaling go beyond mere overthinking and manifest through three unique, domain-specific bottlenecks.</p><p>First, text-centric scaling limits encounter a fundamentally different bottleneck in multimodal clinical environments. While prior work focuses on finding the optimal token length to balance reasoning depth and diversity, our evaluation of VLMs shows that the issue is not about optimal length but rather a structural inability to integrate complex visual clues. Simple token expansion fails to trigger meaningful multimodal reasoning, indicating that safe medical AI deployment requires fundamental vision-language alignment improvements rather than just optimizing computational scaling.</p><p>Second, we identify a domain-specific disparity in scaling efficiency between qualitative clinical and quantitative procedural logic. Unlike general studies that evaluate scalability based on parameter size or general reasoning datasets, we demonstrate that medically fine-tuned LLMs excel in text-based medical QA but exhibit degraded scaling efficiency on clinical calculation tasks compared to general-domain models. This disparity does not necessarily imply that medical fine-tuning inherently impairs reasoning capabilities. Instead, it suggests that current medical alignment heavily prioritizes qualitative clinical knowledge over the general numerical and procedural reasoning captured by calculation benchmarks. Consequently, test-time scaling strategies behave inconsistently depending on whether the clinical task relies on deep domain-specific expertise or general numerical computation, an insight unique to the medical domain.</p><p>Third, our work uncovers a unique cognitive vulnerability to perceived clinical authority. Standard robustness evaluations in prior test-time scaling typically focus on a model&#x2019;s ability to self-correct against random noise or generic misleading logic. In contrast, we demonstrate that when models are confronted with deceptive hints framed specifically as expert physician opinions, they readily abandon their correct reasoning pathways. Under this hierarchical clinical pressure, even optimal scaling strategies fail to restore baseline performance, which represents a distinct vulnerability that differentiates our work from general-domain robustness findings.</p></sec><sec id="s4-3"><title>Practical Considerations for Clinical Deployment</title><p>While our experimental design controls for the token budget to evaluate the cognitive limits of reasoning length, it is important to acknowledge that token count does not equate to equal computational cost across different model architectures. Generating the same number of tokens with a large-scale model consumes substantially more compute (floating point operations per second), energy, and financial resources than a smaller model. Therefore, evaluating deployment feasibility requires balancing reasoning performance against strict computational constraints.</p><p>Furthermore, different test-time scaling strategies exhibit distinct cost-performance trade-offs. As observed in our experiments, parallel scaling can be implemented efficiently using batched inference, where model weights are loaded only once. This minimizes the increase in GPU memory footprint to just the required key-value cache and significantly reduces wall-clock time, making it highly suitable for time-sensitive environments such as emergency departments. Conversely, sequential scaling inherently increases latency due to its serial nature and expanding context length, limiting its use to nonurgent, complex diagnostic scenarios. Ultimately, health care institutions must carefully weigh these factors. They should choose whether to deploy a massive model or to apply efficient parallel scaling on a smaller, medically aligned model based on their specific latency requirements and computational budgets.</p></sec><sec id="s4-4"><title>Limitations</title><p>Although we conducted a broad investigation of test-time scaling in the medical domain, several limitations remain. First, our study focused on scaling strategies such as increasing the token budget, iterative self-revision via sequential scaling, and generating multiple responses in parallel. However, other promising strategies, such as using trained verifiers to select the most accurate response among candidates, were not explored. In addition, compared to LLMs, VLMs require more comprehensive evaluation under test-time scaling. While our results show that test-time scaling can improve robustness to user-driven factors, it was not sufficient to fully restore performance to baseline levels observed without such perturbations. Future work should investigate more advanced or hybrid test-time scaling techniques that can further mitigate the impact of user-driven factors. Finally, our experiments were limited to open-source LLMs, while proprietary models such as GPT (OpenAI), Claude (Anthropic, PBC), or Gemini (Google LLC) were not included. Extending test-time scaling evaluations to these models represents an important direction for future research in the medical domain.</p><p>A notable limitation of our study is that the evaluation relies on specific benchmark datasets, which serve as finite samples of the much broader population of real-world medical questions. Consequently, the reported accuracy metrics carry inherent statistical uncertainty. In scenarios where performance differences across models or scaling strategies are marginal, identifying a single optimal model based solely on empirical accuracy may not fully reflect their true generalization ability. Future research evaluating LLMs in health care should incorporate rigorous statistical frameworks to quantify this uncertainty. Using methodologies to construct confidence sets of potential best performers [<xref ref-type="bibr" rid="ref61">61</xref>-<xref ref-type="bibr" rid="ref63">63</xref>] would provide a more robust and statistically sound approach to model selection when a clear single winner is not evident.</p><p>Additionally, as with most current LLM evaluations, we cannot completely rule out the possibility of data contamination, where public benchmark questions might have been included in the base models&#x2019; pretraining corpora. However, the core focus of this study is the relative performance gain achieved through test-time scaling rather than absolute baseline accuracy. If models were merely retrieving memorized answers, increasing the token budget or applying sequential scaling would not yield the significant accuracy improvements we observed. Furthermore, our perturbation experiments using user-driven factors demonstrate that models engage in active, dynamic reasoning rather than static memory retrieval. Nonetheless, future studies using completely private or recently curated clinical datasets will be essential to further validate the extent of test-time scaling benefits free from any potential memorization bias.</p></sec><sec id="s4-5"><title>Conclusions</title><p>In this paper, we conducted a comprehensive investigation of test-time scaling in the medical domain across various types of LLMs and VLMs. Our experiments demonstrate that longer reasoning is not universally beneficial across all medical tasks. For simpler tasks, extensive reasoning is unnecessary, and models that produce concise reasoning traces offer better computational efficiency. In contrast, for complex and challenging tasks, generating longer chain-of-thought reasoning leads to better performance, with models capable of producing extended reasoning traces outperforming their counterparts. Additionally, we observed that medical LLMs tend to perform worse on medical calculation tasks compared to general domain LLMs. This trend likely reflects that current medical fine-tuning practices heavily focus on qualitative textual medical knowledge, meaning these models are less optimized for the procedural, calculation-heavy demands captured by such benchmarks rather than enduring an actual impairment of reasoning capabilities.</p><p>Furthermore, we examined whether sequential or parallel test-time scaling provides greater benefits. Our findings indicate that the optimal test-time scaling strategy varies depending on both the model type and the difficulty of the task. Additionally, we found that selecting an appropriate scaling strategy is crucial for enhancing model robustness against user-driven factors, such as misleading or biased information in the prompts. Notably, the impact of these user-driven factors becomes more pronounced as task difficulty increases, emphasizing the importance of minimizing perturbations by providing only neutral and essential information for complex tasks. A summary of our findings is presented in <xref ref-type="table" rid="table3">Tables 3</xref> and <xref ref-type="table" rid="table4">4</xref>.</p><p>Beyond benchmark performance, these findings hold broader implications for the real-world deployment of medical AI. As health care institutions increasingly integrate AI into clinical workflows, treating inference compute as a static resource is no longer viable or safe. Our study highlights that ensuring patient safety, diagnostic accuracy, and operational efficiency requires a dynamic approach to model deployment. Computational strategies must be carefully tailored to the specific urgency and complexity of the medical scenario. Ultimately, understanding when and how to appropriately scale reasoning at test time is a crucial foundational step toward developing clinical AI systems that are not only computationally efficient but also robust, resilient, and trustworthy in high-stakes health care environments.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Recommended test-time scaling strategies based on task difficulty in the medical domain.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Difficulty</td><td align="left" valign="bottom">Recommendation</td></tr></thead><tbody><tr><td align="left" valign="top">Easy</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Prefer parallel scaling over simply increasing the token budget or using iterative sequential scaling</p></list-item><list-item><p>Avoid user-driven factors that use a definitive tone and reflect expert-level input</p></list-item></list></td></tr><tr><td align="left" valign="top">Intermediate</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Adapt the scaling strategy based on the model&#x2019;s token usage pattern, combining sequential and parallel scaling as appropriate</p></list-item><list-item><p>Avoid user-driven factors that use a definitive tone</p></list-item></list></td></tr><tr><td align="left" valign="top">Difficult</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Apply iterative sequential scaling for question-answering tasks to enhance step-by-step reasoning</p></list-item><list-item><p>Increase the token budget for calculation tasks that require extended reasoning</p></list-item><list-item><p>Avoid all types of user-driven factors and provide only neutral and essential information</p></list-item></list></td></tr></tbody></table></table-wrap><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Recommended test-time scaling based on model types in the medical domain.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model type</td><td align="left" valign="bottom">Strength</td><td align="left" valign="bottom">Weakness</td><td align="left" valign="bottom">Recommendation</td></tr></thead><tbody><tr><td align="left" valign="top">General, nonreasoning</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Performs well on medical calculation tasks</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Limited medical knowledge</p></list-item><list-item><p>Relatively low performance on reasoning-intensive medical question answering tasks</p></list-item><list-item><p>Use a small number of tokens for reasoning</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Use primarily for medical calculation tasks</p></list-item><list-item><p>Prefer parallel scaling over other strategies</p></list-item></list></td></tr><tr><td align="left" valign="top">General, reasoning</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Performs well on medical calculation tasks</p></list-item><list-item><p>High performance on reasoning-intensive tasks</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Limited medical knowledge</p></list-item><list-item><p>Relatively low performance on reasoning-intensive medical question answering tasks</p></list-item><list-item><p>Underperform when using small model sizes</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Use primarily for medical calculation tasks</p></list-item><list-item><p>Adjust scaling strategy based on the model&#x2019;s token usage pattern and task difficulty, combining sequential and parallel scaling as needed</p></list-item><list-item><p>Use sufficiently large models (over 30B parameters)</p></list-item></list></td></tr><tr><td align="left" valign="top">Medical, nonreasoning</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Performs well on medical question-answering tasks</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Relatively low performance on reasoning-intensive medical calculation tasks</p></list-item><list-item><p>Use a small number of tokens for reasoning</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Use primarily for medical question answering tasks</p></list-item><list-item><p>Prefer parallel scaling over other strategies</p></list-item></list></td></tr><tr><td align="left" valign="top">Medical, reasoning</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Performs well on medical question-answering tasks</p></list-item><list-item><p>High performance on reasoning-intensive tasks</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Relatively low performance on reasoning-intensive medical calculation tasks</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Use primarily for medical question answering tasks</p></list-item><list-item><p>Adjust scaling strategy based on the model&#x2019;s token usage pattern and task difficulty, combining sequential and parallel scaling as needed</p></list-item></list></td></tr></tbody></table></table-wrap></sec></sec></body><back><ack><p>We acknowledge MID (Medical Illustration and Design), as a member of the Medical Research Support Services of Yonsei University College of Medicine, for providing excellent support with medical illustration. We acknowledge the use of generative AI tools (eg, Gemini Pro [Google LLC]) to assist with the refinement, correction, editing, and formatting of this paper to enhance linguistic clarity. No scientific content was generated by these models. All concepts, analyses, and conclusions remain the original work of the authors.</p></ack><notes><sec><title>Funding</title><p>This research was supported by the Bio&#x0026;Medical Technology Development Program of the National Research Foundation (NRF) funded by the Korean government (MSIT; RS-2025-02263045), a grant of the Korean ARPA-H Project through the Korea Health Industry Development Institute (KHIDI), funded by the Ministry of Health &#x0026; Welfare, Republic of Korea (RS-2025-25455095), Institute of Information and Communications Technology Planning &#x0026; Evaluation (IITP) grant funded by the Korea government (Ministry of Science and ICT [information and communication technology]; RS-2026-25518389), the InnoCORE program of the Ministry of Science and ICT (26-InnoCORE-02), a grant from the Institute for AI and Social Innovation at Yonsei University (2025-22-0484), Severance Hospital Research fund for Clinical excellence (SHRC; C-2025-0033), the National Research Foundation of Korea (NRF) grant funded by the Korea government (MSIT; RS-2026-25485764), and a grant of the Korea Health Technology R&#x0026;D (research and development) Project through the Korea Health Industry Development Institute (KHIDI), funded by the Ministry of Health &#x0026; Welfare, Republic of Korea (RS-2025-02263802).The funder had no involvement in this study&#x2019;s design, data collection, analysis, interpretation, or the writing of this paper.</p></sec><sec><title>Data Availability</title><p>All datasets used in this study are publicly available.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: GO, SP, BHK</p><p>Data curation: GO</p><p>Formal analysis: GO</p><p>Funding acquisition: SP, BHK</p><p>Investigation: GO (lead), SK (supporting)</p><p>Methodology: GO (lead), SP (supporting), BHK (supporting)</p><p>Project administration: SP, BHK</p><p>Resources: SP, BHK</p><p>Software: GO</p><p>Supervision: SP, BHK</p><p>Validation: GO (lead), SK (supporting)</p><p>Visualization: GO</p><p>Writing &#x2013; original draft: GO (lead), SK (supporting)</p><p>Writing &#x2013; review &#x0026; editing: GO, SP (equal), BHK (equal)</p><p>BHK and SP are co-corresponding authors of this work. Correspondence regarding this paper may be addressed to either BHK (egyptdj@yonsei.ac.kr) or SP (depecher@yuhs.ac).</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CLIMB</term><def><p>clinical large-scale integrative multimodal benchmark</p></def></def-item><def-item><term id="abb2">ECG</term><def><p>electrocardiogram</p></def></def-item><def-item><term id="abb3">LLaVA</term><def><p>Large Language and Vision Assistant</p></def></def-item><def-item><term id="abb4">LLaVA-CoT</term><def><p>Large Language and Vision Assistant&#x2013;Chain-of-Thought</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">QA</term><def><p>question answering</p></def></def-item><def-item><term id="abb7">QVQ</term><def><p>Qwen With Vision and Questions</p></def></def-item><def-item><term id="abb8">RL</term><def><p>reinforcement learning</p></def></def-item><def-item><term id="abb9">SFT</term><def><p>supervised fine-tuning</p></def></def-item><def-item><term id="abb10">USMLE </term><def><p>United States Medical Licensing Examination</p></def></def-item><def-item><term id="abb11">VLM</term><def><p>vision-language model</p></def></def-item><def-item><term id="abb12">VQA</term><def><p>visual question answering</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mann</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ryder</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Language models are few-shot learners</article-title><access-date>2026-07-01</access-date><conf-name>NIPS&#x2019;20: Proceedings of the 34th International Conference on Neural Information Processing Systems</conf-name><conf-date>Dec 6-12, 2020</conf-date><conf-loc>Vancouver, British Columbia, Canada</conf-loc><fpage>1877</fpage><lpage>1901</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf">https://proceedings.neurips.cc/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Achiam</surname><given-names>J</given-names> </name><name name-style="western"><surname>Adler</surname><given-names>S</given-names> </name><name name-style="western"><surname>Agarwal</surname><given-names>S</given-names> </name><collab>OpenAI</collab><etal/></person-group><article-title>GPT-4 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 15, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.08774</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Chiang</surname><given-names>WL</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Vicuna: an open-source chatbot impressing GPT-4 with 90%* ChatGPT quality</article-title><source>LMSYS</source><year>2023</year><month>03</month><day>30</day><access-date>2026-07-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.lmsys.org/blog/2023-03-30-vicuna/">https://www.lmsys.org/blog/2023-03-30-vicuna/</ext-link></comment></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Touvron</surname><given-names>H</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Stone</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Llama 2: open foundation and fine-tuned chat models</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 18, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2307.09288</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Grattafiori</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dubey</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jauhri</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The llama 3 herd of models</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 31, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2407.21783</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hurst</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lerer</surname><given-names>A</given-names> </name><name name-style="western"><surname>Goucher</surname><given-names>AP</given-names> </name><collab>OpenAI</collab><etal/></person-group><article-title>GPT-4o system card</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 25, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2410.21276</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Jaech</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kalai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lerer</surname><given-names>A</given-names> </name><etal/></person-group><article-title>OpenAI o1 system card</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 21, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.16720</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Qwen</surname><given-names>AY</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Qwen2.5 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 19, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.15115</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning</article-title><source>Nature</source><year>2025</year><month>09</month><volume>645</volume><issue>8081</issue><fpage>633</fpage><lpage>638</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-09422-z</pub-id><pub-id pub-id-type="medline">40962978</pub-id></nlm-citation></ref><ref id="ref10"><label>5</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Alayrac</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Donahue</surname><given-names>J</given-names> </name><name name-style="western"><surname>Luc</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Flamingo: A Visual Language Model for Few-shot Learning</article-title><conf-name>36th Conference on Neural Information Processing Systems (NeurIPS 2022)</conf-name><conf-date>Nov 28 to Dec 9, 2022</conf-date><conf-loc>New Orleans, LA</conf-loc><fpage>23716</fpage><lpage>23736</lpage><pub-id pub-id-type="doi">10.52202/068431-1723</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>D</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>C</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Hoi</surname><given-names>S</given-names> </name></person-group><article-title>BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation</article-title><access-date>2026-07-01</access-date><conf-name>International Conference on Machine Learning</conf-name><conf-date>Jul 17-23, 2022</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v162/li22n/li22n.pdf">https://proceedings.mlr.press/v162/li22n/li22n.pdf</ext-link></comment></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>D</given-names> </name><name name-style="western"><surname>Savarese</surname><given-names>S</given-names> </name><collab>editors</collab></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Hoi</surname><given-names>S</given-names> </name></person-group><article-title>BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models</article-title><access-date>2026-07-01</access-date><conf-name>International Conference on Machine Learning</conf-name><conf-date>Jul 23-29, 2023</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v202/li23q/li23q.pdf">https://proceedings.mlr.press/v202/li23q/li23q.pdf</ext-link></comment></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>YJ</given-names> </name></person-group><article-title>Visual Instruction Tuning</article-title><conf-name>Advances in Neural Information Processing Systems 36</conf-name><conf-date>Dec 10-16, 2023</conf-date><conf-loc>New Orleans, LA</conf-loc><fpage>34892</fpage><lpage>34916</lpage><pub-id pub-id-type="doi">10.52202/075280-1516</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="other"><person-group person-group-type="author"><collab>Gemini Team Google</collab><name name-style="western"><surname>Georgiev</surname><given-names>P</given-names> </name><name name-style="western"><surname>Lei</surname><given-names>VI</given-names> </name><name name-style="western"><surname>Burnell</surname><given-names>R</given-names> </name></person-group><article-title>Gemini 1.5: unlocking multimodal understanding across millions of tokens of context</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 8, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2403.05530</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bai</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Qwen2-VL: enhancing vision-language model&#x2019;s perception of the world at any resolution</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 18, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2409.12191</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Xue</surname><given-names>L</given-names> </name><name name-style="western"><surname>Shu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Awadalla</surname><given-names>A</given-names> </name><etal/></person-group><article-title>XGen-MM (BLIP-3): a family of open large multimodal models</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 16, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2408.08872</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Muennighoff</surname><given-names>N</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>W</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Hajishirzi</surname><given-names>H</given-names> </name></person-group><article-title>S1: simple test-time scaling</article-title><conf-name>Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 4-9, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.emnlp-main.1025</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Snell</surname><given-names>CV</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>A</given-names> </name></person-group><article-title>Scaling LLM test-time compute optimally can be more effective than scaling parameters for reasoning</article-title><conf-name>The Thirteenth International Conference on Learning Representations</conf-name><conf-date>Apr 24-28, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.emnlp-main</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>Y</given-names> </name><collab>editors</collab></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Wei</surname><given-names>F</given-names> </name></person-group><article-title>Towards thinking-optimal scaling of test-time compute for LLM reasoning</article-title><access-date>2026-07-01</access-date><conf-name>The Thirty-ninth Annual Conference on Neural Information Processing Systems</conf-name><conf-date>Dec 2-7, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2025/hash/3e22bea3b170f4c2aebb9c48d98ae64d-Abstract-Conference.html">https://proceedings.neurips.cc/paper_files/paper/2025/hash/3e22bea3b170f4c2aebb9c48d98ae64d-Abstract-Conference.html</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zeng</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>X</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Qiu</surname><given-names>X</given-names> </name></person-group><article-title>Revisiting the test-time scaling of o1-like models: do they truly possess test-time scaling capabilities?</article-title><conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1)</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.232</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Schulman</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wolski</surname><given-names>F</given-names> </name><name name-style="western"><surname>Dhariwal</surname><given-names>P</given-names> </name><name name-style="western"><surname>Radford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Klimov</surname><given-names>O</given-names> </name></person-group><article-title>Proximal policy optimization algorithms</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 20, 2017</comment><pub-id pub-id-type="doi">10.48550/arXiv.1707.06347</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Rafailov</surname><given-names>R</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mitchell</surname><given-names>E</given-names> </name><name name-style="western"><surname>Manning</surname><given-names>CD</given-names> </name><name name-style="western"><surname>Ermon</surname><given-names>S</given-names> </name><name name-style="western"><surname>Finn</surname><given-names>C</given-names> </name></person-group><article-title>Direct preference optimization: your language model is secretly a reward model</article-title><conf-name>NIPS &#x2019;23: Proceedings of the 37th International Conference on Neural Information Processing Systems</conf-name><conf-date>Dec 10-16, 2023</conf-date><conf-loc>New Orleans, LA</conf-loc><fpage>53728</fpage><lpage>53741</lpage><pub-id pub-id-type="doi">10.52202/075280-2338</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Shao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>DeepSeekMath: pushing the limits of mathematical reasoning in open language models</article-title><source>arXiv</source><comment>Preprint posted online on  Feb, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2402.03300</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moon</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>H</given-names> </name><name name-style="western"><surname>Shin</surname><given-names>W</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Choi</surname><given-names>E</given-names> </name></person-group><article-title>Multi-modal understanding and generation for medical images and text via vision-language pre-training</article-title><source>IEEE J Biomed Health Inf</source><year>2022</year><month>12</month><volume>26</volume><issue>12</issue><fpage>6070</fpage><lpage>6080</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2022.3207502</pub-id><pub-id pub-id-type="medline">36121943</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><etal/></person-group><article-title>LLaVA-med: training a large language-and-vision assistant for biomedicine in one day</article-title><conf-name>37th Conference on Neural Information Processing Systems (NeurIPS 2023) Track on Datasets and Benchmarks</conf-name><conf-date>Dec 10-16, 2023</conf-date><conf-loc>New Orleans LA</conf-loc><fpage>28541</fpage><lpage>28564</lpage><pub-id pub-id-type="doi">10.52202/075280-1240</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>ES</given-names> </name><name name-style="western"><surname>Shin</surname><given-names>KS</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Ye</surname><given-names>JC</given-names> </name></person-group><article-title>Self-supervised multi-modal training from uncurated images and reports enables monitoring AI in radiology</article-title><source>Med Image Anal</source><year>2024</year><month>01</month><volume>91</volume><fpage>103021</fpage><pub-id pub-id-type="doi">10.1016/j.media.2023.103021</pub-id><pub-id pub-id-type="medline">37952385</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hua</surname><given-names>E</given-names> </name><etal/></person-group><article-title>UltraMedical: building specialized generalists in biomedicine</article-title><conf-name>38th Conference on Neural Information Processing Systems (NeurIPS 2024) Track on Datasets and Benchmarks</conf-name><conf-date>Dec 10-15, 2024</conf-date><conf-loc>Vancouver, British Columbia, Canada</conf-loc><fpage>26045</fpage><lpage>26081</lpage><pub-id pub-id-type="doi">10.52202/079017-0819</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>R</given-names> </name><name name-style="western"><surname>Adhikarla</surname><given-names>E</given-names> </name><etal/></person-group><article-title>A generalist vision-language foundation model for diverse biomedical tasks</article-title><source>Nat Med</source><year>2024</year><month>11</month><volume>30</volume><issue>11</issue><fpage>3129</fpage><lpage>3141</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03185-2</pub-id><pub-id pub-id-type="medline">39112796</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Towards medical complex reasoning with llms through medical verifiable problems</article-title><conf-name>Findings of the Association for Computational Linguistics: ACL 2025</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.findings-acl.751</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name></person-group><article-title>MedS: towards medical slow thinking with self-evolved soft dual-sided process supervision</article-title><conf-name>The Fortieth AAAI Conference on Artificial Intelligence (AAAI-26)</conf-name><conf-date>Feb 25 to Mar 4, 2025</conf-date><conf-loc>Philadelphia,PA</conf-loc><fpage>31319</fpage><lpage>31327</lpage><pub-id pub-id-type="doi">10.1609/aaai.v40i37.40395</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dai</surname><given-names>W</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ekbote</surname><given-names>C</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>PP</given-names> </name></person-group><article-title>QoQ-med: building multimodal clinical foundation models with domain-aware GRPO training</article-title><access-date>2026-07-02</access-date><conf-name>39th Conference on Neural Information Processing Systems (NeurIPS 2025)</conf-name><conf-date>Dec 2-7, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2025/file/359ffa88712bd688963a0ca641d8330b-Paper-Conference.pdf">https://proceedings.neurips.cc/paper_files/paper/2025/file/359ffa88712bd688963a0ca641d8330b-Paper-Conference.pdf</ext-link></comment></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lai</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhong</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Med-R1: reinforcement learning for generalizable medical reasoning in vision-language models</article-title><source>IEEE Trans Med Imaging</source><year>2026</year><month>06</month><volume>45</volume><issue>6</issue><fpage>2727</fpage><lpage>2737</lpage><pub-id pub-id-type="doi">10.1109/TMI.2026.3661001</pub-id><pub-id pub-id-type="medline">41632678</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>J</given-names> </name><etal/></person-group><article-title>MedVLM-R1: incentivizing medical reasoning capability of vision-language models (vlms) via reinforcement learning</article-title><conf-name>International Conference on Medical Image Computing and Computer-Assisted Intervention</conf-name><conf-date>Sep 23-27, 2025</conf-date><pub-id pub-id-type="doi">10.1007/978-3-032-04981-0_32</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name></person-group><article-title>M1: unleash the potential of test-time scaling for medical reasoning with large language models. the second workshop on genai for health: potential, trust, and policy compliance</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 1, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.00869</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Geng</surname><given-names>G</given-names> </name><name name-style="western"><surname>Hua</surname><given-names>S</given-names> </name><etal/></person-group><article-title>O1 replication journey--part 3: inference-time scaling for medical reasoning</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 11, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.06458</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Balachandran</surname><given-names>V</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Inference-time scaling for complex tasks: where we stand and what lies ahead</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 31, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.00294</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kaya</surname><given-names>MO</given-names> </name><name name-style="western"><surname>Elliott</surname><given-names>D</given-names> </name><name name-style="western"><surname>Papadopoulos</surname><given-names>DP</given-names> </name></person-group><article-title>Efficient test-time scaling for small vision-language models</article-title><access-date>2026-07-04</access-date><conf-name>The Fourteenth International Conference on Learning Representations</conf-name><conf-date>Apr 23-27, 2026</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://monurcan.github.io/efficient_test_time_scaling/">https://monurcan.github.io/efficient_test_time_scaling/</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gui</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ouyang</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Towards injecting medical visual knowledge into multimodal llms at scale</article-title><conf-name>Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 12-16, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.418</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Sellergren</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kazemzadeh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jaroensri</surname><given-names>T</given-names> </name><etal/></person-group><article-title>MedGemma technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 7, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2507.05201</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Dhingra</surname><given-names>B</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>W</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>X</given-names> </name></person-group><article-title>PubMedQA: a dataset for biomedical research question answering</article-title><conf-name>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</conf-name><conf-date>Nov 3-7, 2019</conf-date><pub-id pub-id-type="doi">10.18653/v1/D19-1259</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Oufattole</surname><given-names>N</given-names> </name><name name-style="western"><surname>Weng</surname><given-names>WH</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Szolovits</surname><given-names>P</given-names> </name></person-group><article-title>What disease does this patient have? A large-scale open domain question answering dataset from medical exams</article-title><source>Appl Sci</source><year>2021</year><volume>11</volume><issue>14</issue><fpage>6421</fpage><pub-id pub-id-type="doi">10.3390/app11146421</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Singla</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Dredze</surname><given-names>M</given-names> </name></person-group><article-title>Benchmarking large language models on answering and explaining challenging medical questions</article-title><conf-name>Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)</conf-name><conf-date>Apr 29 to May 4, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.naacl-long.182</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zuo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Qu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Hua</surname><given-names>E</given-names> </name></person-group><article-title>MedXpertQA: benchmarking expert-level medical reasoning and understanding</article-title><access-date>2026-07-02</access-date><conf-name>Proceedings of the 42nd International Conference on Machine Learning</conf-name><conf-date>Jul 13-19, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=IyVcxU0RKI">https://openreview.net/pdf?id=IyVcxU0RKI</ext-link></comment></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Khandekar</surname><given-names>N</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>G</given-names> </name><etal/></person-group><article-title>MedCalc-bench: evaluating large language models for medical calculations</article-title><conf-name>38th Conference on Neural Information Processing Systems (NeurIPS 2024) Track on Datasets and Benchmarks</conf-name><conf-date>Dec 10-15, 2024</conf-date><conf-loc>Vancouver, British Columbia, Canada</conf-loc><fpage>84730</fpage><lpage>84745</lpage><pub-id pub-id-type="doi">10.52202/079017-2690</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lim</surname><given-names>KH</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>U</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Susceptibility of large language models to user-driven factors in medical queries</article-title><source>J Healthcare Inf Res</source><year>2026</year><month>06</month><volume>10</volume><issue>2</issue><fpage>498</fpage><lpage>522</lpage><pub-id pub-id-type="doi">10.1007/s41666-025-00218-4</pub-id><pub-id pub-id-type="medline">42110657</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Q</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Qiao</surname><given-names>Y</given-names> </name></person-group><article-title>OmniMedVQA: a new large-scale comprehensive evaluation benchmark for medical LVLM</article-title><conf-name>2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name><conf-date>Jun 16-22, 2024</conf-date><pub-id pub-id-type="doi">10.1109/CVPR52733.2024.02093</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ouyang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Training language models to follow instructions with human feedback</article-title><conf-name>36th Conference on Neural Information Processing Systems (NeurIPS 2022)</conf-name><conf-date>Nov 28 to Dec 9, 2022</conf-date><conf-loc>New Orleans, LA</conf-loc><fpage>27730</fpage><lpage>27744</lpage><pub-id pub-id-type="doi">10.52202/068431-2011</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Li</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Qwen3 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  May 14, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.09388</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Agarwal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ahmad</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ai</surname><given-names>J</given-names> </name><collab>OpenAI</collab><etal/></person-group><article-title>Gpt-oss-120b &#x0026; gpt-oss-20b model card</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 8, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2508.10925</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ethayarajh</surname><given-names>K</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Muennighoff</surname><given-names>N</given-names> </name><name name-style="western"><surname>Jurafsky</surname><given-names>D</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Kiela</surname><given-names>D</given-names> </name></person-group><article-title>KTO: Model alignment as prospect theoretic optimization</article-title><access-date>2026-07-02</access-date><conf-name>Proceedings of the 41 st International Conference on Machine Learning</conf-name><conf-date>Jul 21-27, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/pdf/2402.01306">https://arxiv.org/pdf/2402.01306</ext-link></comment></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Bai</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Qwen2.5-VL technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 19, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2502.13923</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Bai</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>R</given-names> </name></person-group><article-title>Qwen3-VL technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 26, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2511.21631</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="other"><person-group person-group-type="author"><collab>Team</collab><name name-style="western"><surname>Kamath</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ferret</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pathak</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Gemma 3 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 25, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2503.19786</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>G</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>P</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>LlaVA-cot: let vision language models reason step-by-step</article-title><conf-name>2025 IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name><conf-date>Oct 19-25, 2025</conf-date><pub-id pub-id-type="doi">10.1109/ICCV51701.2025.00202</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="web"><person-group person-group-type="author"><collab>QwenTeam</collab></person-group><article-title>QVQ: to see the world with wisdom</article-title><year>2024</year><month>12</month><day>24</day><access-date>2026-07-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://qwen.ai/blog?id=qvq-72b-preview">https://qwen.ai/blog?id=qvq-72b-preview</ext-link></comment></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dai</surname><given-names>W</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>P</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>M</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Cui</surname><given-names>H</given-names> </name></person-group><article-title>CLIMB: data foundations for large scale multimodal clinical foundation models</article-title><access-date>2026-07-02</access-date><conf-name>Forty-second International Conference on Machine Learning</conf-name><conf-date>Jul 13-19, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=TcvjOSePic">https://openreview.net/pdf?id=TcvjOSePic</ext-link></comment></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>C</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>XH</given-names> </name><etal/></person-group><article-title>Soft adaptive policy optimization</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 25, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2511.20347</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Young</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>B</given-names> </name><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Yi: open foundation models by 01.AI</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 7, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2403.04652</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Benton</surname><given-names>J</given-names> </name><name name-style="western"><surname>Radhakrishnan</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Reasoning models don&#x2019;t always say what they think</article-title><source>arXiv</source><comment>Preprint posted online on  May 8, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.05410</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Thapa</surname><given-names>R</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Disentangling reasoning and knowledge in medical large language models</article-title><source>arXiv</source><comment>Preprint posted online on  May 16, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.11462</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hansen</surname><given-names>PR</given-names> </name><name name-style="western"><surname>Lunde</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nason</surname><given-names>JM</given-names> </name></person-group><article-title>The model confidence set</article-title><source>Econometrica</source><year>2011</year><volume>79</volume><issue>2</issue><fpage>453</fpage><lpage>497</lpage><pub-id pub-id-type="doi">10.3982/ECTA5771</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lei</surname><given-names>J</given-names> </name></person-group><article-title>Winners with confidence: discrete argmin inference with an application to model selection</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 4, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2408.02060</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>I</given-names> </name><name name-style="western"><surname>Ramdas</surname><given-names>A</given-names> </name></person-group><article-title>Locally minimax optimal confidence sets for the best model</article-title><comment>Preprint posted online on  Mar 27, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2503.21639</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Prompts for large language models or vision-language models, examples of responses of QVQ (Qwen With Vision and Questions), and results of ablation studies.</p><media xlink:href="jmir_v28i1e90693_app1.docx" xlink:title="DOCX File, 4553 KB"/></supplementary-material></app-group></back></article>