<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e97221</article-id><article-id pub-id-type="doi">10.2196/97221</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>The Alignment Paradox of Medical Large Language Models in Infertility Care: Decoupling Algorithmic Improvement From Clinical Decision-Making Quality</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Liu</surname><given-names>Dou</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Long</surname><given-names>Ying</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zuoqiu</surname><given-names>Sophia</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Di</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Kang</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Xie</surname><given-names>Kaipeng</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yang</surname><given-names>Runze</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lin</surname><given-names>Yiting</given-names></name><degrees>MBBS</degrees><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Hanyi</given-names></name><degrees>MBBS</degrees><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name name-style="western"><surname>Yin</surname><given-names>Rong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Tang</surname><given-names>Tian</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Obstetrics and Gynecology, West China Second University Hospital of Sichuan University</institution><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Industrial and Operations Engineering, University of Michigan</institution><addr-line>Ann Arbor</addr-line><addr-line>MI</addr-line><country>United States</country></aff><aff id="aff3"><institution>Key Laboratory of Birth Defects and Related Diseases of Women and Children, Sichuan University</institution><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><aff id="aff4"><institution>Reproductive Medical Center, Department of Obstetrics and Gynecology, West China Second University Hospital of Sichuan University</institution><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><aff id="aff5"><institution>Department of Industrial Engineering, Sichuan University</institution><addr-line>No.37, Guoxue Lane, Wuhou District</addr-line><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><aff id="aff6"><institution>West China Biomedical Big Data Center, West China Hospital, Sichuan University</institution><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><aff id="aff7"><institution>Med-X Center for Informatics, Sichuan University</institution><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><aff id="aff8"><institution>West China School of Medicine, Sichuan University</institution><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Lv</surname><given-names>Huasheng</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Wang</surname><given-names>Yijie</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Rong Yin, PhD, Department of Industrial Engineering, Sichuan University, No.37, Guoxue Lane, Wuhou District, Chengdu, Sichuan, China, 86 17208267909; <email>rong.yin@scupi.cn</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e97221</elocation-id><history><date date-type="received"><day>05</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>14</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>18</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Dou Liu, Ying Long, Sophia Zuoqiu, Di Liu, Kang Li, Kaipeng Xie, Runze Yang, Yiting Lin, Hanyi Liu, Rong Yin, Tian Tang. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 21.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e97221"/><abstract><sec><title>Background</title><p>Large language models (LLMs) have been proposed as decision-support tools in assisted reproductive technology (ART), but it remains unclear whether posttraining alignment strategies translate into clinically acceptable decision support. Outcome-based benchmarks may reward token-level correctness while overlooking the reasoning quality that clinicians rely on.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate whether 4 mainstream alignment paradigms for medical LLMs (supervised fine-tuning [SFT], direct preference optimization [DPO], group relative policy optimization [GRPO], and in-context learning [ICL]) produce comparable algorithmic and clinical alignment (SFT vs GRPO) when used for infertility diagnosis and treatment planning.</p></sec><sec sec-type="methods"><title>Methods</title><p>This retrospective single-center study used 8201 deidentified electronic health records from West China Second University Hospital, collected between January 2020 and December 2022 (mean age 31.79, SD 4.63 y). All 4 strategies were built on a shared open-source biomedical backbone. Evaluation comprised the following 2 tiers: (1) automatic field-level metrics (accuracy, macro-<italic>F</italic><sub>1</sub>, and mean absolute error [MAE]) on 5 structured decision fields (infertility type, initial diagnosis, ART strategy, controlled ovarian stimulation [COS] regimen, and gonadotropin starting dose); and (2) blinded independent expert review by 2 reproductive medicine specialists on 100 paired cases across 4 clinical dimensions (reasoning capability, diagnostic accuracy, treatment feasibility, and hallucination). Automatic field-level evaluation included all 4 strategies, whereas blinded expert review was restricted to the SFT versus GRPO contrast.</p></sec><sec sec-type="results"><title>Results</title><p>GRPO achieved the highest average automatic performance (eg, infertility type accuracy=92.57%, COS regimen accuracy=62.36%, ART strategy accuracy=76.49%, and gonadotropin dose MAE=44.94). However, in blinded expert review, the conservative SFT baseline showed directionally higher expert ratings than GRPO on reasoning capability and treatment feasibility; diagnostic-accuracy differences were not significant. In the 3-way best-response comparison including the original physician-charted plan, the SFT baseline was selected as the best response in 51.2% (102.3/200) of cases compared with 26.2% (52.4/200) for GRPO, and 22.6% (45.3/200) for the charted plan. This result reflects preference within the standardized review format, not evidence that model-generated decisions are clinically superior to physician decision-making. Hallucination rates were 15% (GRPO) and 18.5% (SFT), indicating that higher automatic performance did not eliminate clinically unsupported content and that neither model is ready for clinical deployment. Subgroup analyses showed GRPO improved <italic>F</italic><sub>1</sub> in in vitro fertilization (IVF) and preimplantation genetic testing (PGT) but decreased <italic>F</italic><sub>1</sub> in intracytoplasmic sperm injection (ICSI) cases, where male-factor information was largely captured only in unstructured fields.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Outcome-based metrics alone are insufficient proxies for clinical utility in ART decision support. Algorithmic improvement and clinical alignment suggest a possible decoupling; a phenomenon we term the alignment paradox. However, because the clinical review was based on 2 reproductive medicine specialists with marginal interrater agreement, the clinical-alignment findings should be interpreted as exploratory. External multicenter validation is required before clinical deployment.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>clinical alignment</kwd><kwd>assisted reproductive technology</kwd><kwd>reproductive medicine</kwd><kwd>infertility treatment</kwd><kwd>reinforcement learning from human feedback</kwd><kwd>clinical reasoning</kwd><kwd>medical artificial intelligence</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Infertility is one global health issue that is experienced by 1 in 6 people at a certain stage in their lives, according to the World Health Organization (WHO) [<xref ref-type="bibr" rid="ref1">1</xref>]. Assisted reproductive technology (ART) is a common clinical pathway for these cases if conventional treatments, such as cycle regulation or ovulation induction, do not work [<xref ref-type="bibr" rid="ref2">2</xref>]. ART needs to consider the complexity of high-dimensional data to make an effective treatment plan. Clinicians must also determine the controlled ovarian stimulation (COS) protocol and the gonadotropin starting dose, which are generally sensitive to the patient&#x2019;s physiological state. Consequently, ART decision-making is a multistage, high-dimensional, and evidence-driven process that is both time-consuming and cognitively demanding. Clinical outcomes are highly dependent on accurate reasoning, where inappropriate treatment selection may have negative outcomes such as reduced cycle success rates, increased financial and emotional burden, and even increased the risk of severe complications such as ovarian hyperstimulation syndrome (OHSS) [<xref ref-type="bibr" rid="ref3">3</xref>]. These decisions rely heavily on individual clinical experience, leading to substantial interphysician and/or intercenter variability. The possibility of training generalization for novice physicians is thus more challenging. Moreover, ART decision-making faces structural challenges due to imbalanced medical resources. For top-tier medical centers that have large patient volumes, inducing severe decision fatigue and potential safety risks among experienced specialists [<xref ref-type="bibr" rid="ref4">4</xref>]. On the other hand, primary care often lacks the specialized reasoning capability required for such high-dimensional decision-making, leading to significant intercenter variability in care quality. This &#x201C;capability-capacity gap&#x201D; calls for the need for expert-level clinical decision support systems. The inherently complex nature of ART makes it both a representative real-world medical decision environment and a sensitive setting. Thus, failures of model alignment are likely to be amplified and clinically consequential, if present.</p><p>Large language models (LLMs) have rapidly advanced across domains, while limited studies address the issues about how their advanced alignment interacts with the hierarchical, high-stakes reasoning in real-world medicine, especially reproductive medicine. Currently, typical alignment methods usually involve various reinforcement learning variants such as reinforcement learning from human feedback (RLHF), group relative policy optimization (GRPO), and direct preference optimization (DPO), which have shown remarkable improvement within some general areas by using verifiable rewards or preference signals to optimize the policy model [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref9">9</xref>]. For example, after adopting the GRPO, the performance of DeepSeek-R1 on math questions significantly outperformed the baseline [<xref ref-type="bibr" rid="ref8">8</xref>]. However, clinical decision-making presents fundamentally different challenges; the reasoning chains are often long and multidimensional, where signals are noisy and unverifiable, and clinicians commonly require explanation quality and the entire plan&#x2019;s feasibility rather than solely focusing on the single field output correctness alone. These properties suggest a structural tension between <italic>algorithmic alignment</italic> and <italic>clinical alignment</italic>, which has not been systematically examined.</p><p>The domain-specific medical LLMs, such as Med-PaLM (Google), MedGemma (Google), and Lingshu (Alibaba), were proven to exhibit profound potential in clinical problems, ranging from text classification to clinical image reports [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref16">16</xref>]. Some models demonstrate capabilities with high performance in the answer correctness on benchmark datasets. Despite their outstanding performance, when these state-of-the-art (SOTA) models are applied to specialized real-world clinical cases to support the realistic diagnosis operation, they seem to encounter some obstacles [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. Although SOTA models' generated responses were preferred to those provided by generalist physicians, they still did not outperform domain-specific specialists, indicating limited utility in addressing highly specialized routine clinical problems [<xref ref-type="bibr" rid="ref19">19</xref>]. Moreover, the untransparent reasoning processes limit their clinical adoption, despite evidence that explainable AI enhances trust and interpretability [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref23">23</xref>]. While supervised fine-tuning (SFT) remains the dominant paradigm in medical model development [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref24">24</xref>], limited data and incomplete supervision motivate increasing reliance on posttraining alignment [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Although RLHF-style optimization is increasingly used, little is known about whether improvements in algorithmic metrics translate into enhanced interpretability, trust, or multistep clinical reasoning. Additionally, few studies have evaluated how different alignment paradigms behave in real-world clinical decision environments. Reasoning quality in these scenarios directly affects patient safety. These alignment strategies mainly optimize for algorithmic rewards and often fail to capture the interpretability and process-oriented reasoning distinct to medical practice. Specifically, standard alignment targets may overlook the feasibility of the treatment plan, resulting in model failures in obtaining the trust necessary for clinical application. A fundamental research gap is whether stronger posttraining alignment produces more clinically aligned models.</p><p>In this study, we propose the following 2 key research questions (RQs):</p><list list-type="bullet"><list-item><p>RQ1: How do SFT and reinforcement-based alignment strategies (eg, GRPO and DPO) suffice to capture the high-dimensional complexity of real-world ART treatment and enhance decision-making?</p></list-item><list-item><p>RQ2 (the paradox): Is there a possible decoupling between &#x201C;algorithmic alignment&#x201D; (ie, reward maximization) and &#x201C;clinical alignment&#x201D; (ie, human trust) in high-risk areas (ie, reproductive medicine)?</p></list-item></list><p>We define the &#x201C;alignment paradox&#x201D; as follows: in clinical decision-support settings, posttraining optimization that demonstrably improves algorithmic performance on structured outcome metrics (ie, algorithmic alignment) may simultaneously degrade physician-perceived reasoning quality and treatment feasibility (ie, clinical alignment). In other words, a model may become more accurate on paper while becoming less trustworthy in practice. This paradox is clinically significant because, in high-stakes fields, such as reproductive medicine, clinicians rely not only on the correctness of the final recommendation but also on the transparency and coherence of the reasoning process that leads to it. If optimization erodes this process, it undermines the very foundation of clinical trust, regardless of outcome-level gains. To address these questions, we systematically compare 4 representative alignment paradigms: SFT, DPO, GRPO, and in-context learning (ICL), using a large-scale real-world dataset of more than 8000 infertility cases. To enhance robustness on challenging and long-tail cases, we construct a hierarchical &#x201C;pyramid&#x201D; dataset that emphasizes diverse difficulty levels and decision structures. Most importantly, we introduce a dual-evaluation framework that integrates automated metric-based assessment with blinded independent doctor-in-the-loop evaluations, in which the physicians, the model identities, and the case-model assignments are all blinded, enabling us to directly examine the relationship between algorithmic optimization and clinician-centered preferences. This framework allows us to compare both alignment strategies&#x2019; performance and the relationship between algorithmic and clinical alignment, which is the core of the alignment paradox that we investigate in this study. The overview of the alignment and evaluation framework is presented in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study workflow for evaluating four alignment paradigms for large language models in assisted reproductive technology decision support. A single-center retrospective dataset of 8201 deidentified electronic health records from the reproductive medical center, West China Second University Hospital, Sichuan University (Chengdu, China; January 2020-December 2022) was used to train, validate, and test supervised fine-tuning (SFT), direct preference optimization (DPO), group relative policy optimization (GRPO), and in-context learning (ICL) variants of a shared biomedical backbone. A pyramid-curated alignment dataset combining general, model-confused, and expert-refined cases was used for post-training. Outputs were evaluated by a dual framework combining automatic field-level metrics and blinded independent expert review by reproductive medicine specialists.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97221_fig01.png"/></fig></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Data Source</title><p>This study includes 19,800 electronic health records (EHRs) data from West China Second University Hospital, spanning from January 2020 to December 2022. After data cleaning, a final dataset of 8201 EHRs was included in this study, with patient mean age of 31.79 (SD 4.63) years. Grouped by ART, there are in total 11 different ART methods, which can be generally assigned to 3 ART generations: in vitro fertilization (IVF), intracytoplasmic sperm injection (ICSI), and preimplantation genetic testing (PGT). Since the WHO has regulated the names of PGT and its subtypes, we convert all the outdated names, such as PGD (now PGT for monogenic disorders [PGT-M]), PGS (now PGT for aneuploidy [PGT-A]), to the stipulated ones. Statistically, there are 71.21% (5840/8201) IVF, including 59.6% (4885/8201) standard IVF, 8.17% (670/8201) short protocol IVF (short-time insemination), 3.48% (285/8201) IVF with donor sperm; 17.92% (1470/8201) ICSI, including 11.57% (949/8201) standard ICSI, 3.68% (302/8201) testicular sperm aspiration (TESA)+ICSI, 1.89% (155/8201) IVF+ICSI, 0.65% (53/8201) ICSI with frozen sperm, and 0.13% (11/8201) ICSI with donor sperm; 10.86% (891/8201) PGT, including 5.12% (420/8201) preimplantation genetic testing for structural rearrangements (PGT-SR), 3.76% (308/8201) PGT-A, and 1.98% (163/8201) PGT-M. For COS regimen, there are 12 categories: gonadotropin-releasing hormone antagonist fixed protocol (antagonist-fixed), luteal short-acting long protocol (luteal-short), gonadotropin-releasing hormone antagonist flexible protocol (antagonist-flex), follicular phase long-acting protocol (long-acting), progestin-primed ovarian stimulation protocol (PPOS), clomiphene citrate (CC) plus gonadotropins (CC+gonadotropin), mild stimulation protocol or direct gonadotropin protocol (mild or direct gonadotropin), luteal phase stimulation protocol (luteal-stim), clomiphene or letrozole combined with gonadotropins (CC/Letro+gonadotropin), conventional ultra-long protocol (ultra-long), modified ultra-long protocol (mod-ulong), short protocol (GnRH [gonadotropin-releasing hormone]-a short protocol; short).</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This study is a retrospective secondary analysis of deidentified EHRs. The study protocol was reviewed and approved by the Ethics Committee of West China Second University Hospital, Sichuan University (approval 2022288). Given the retrospective design and the use of fully deidentified records, the Ethics Committee waived the requirement for additional informed consent for this secondary analysis; the original general informed consent obtained at the time of clinical care, together with the institutional policy on secondary research use of deidentified data, was determined to be sufficient.</p><p>All EHR data were fully deidentified before extraction. The deidentification procedure followed HIPAA (Health Insurance Portability and Accountability Act) Safe Harbor standards, including the removal of all direct identifiers and quasi-identifiers (eg, names, medical record numbers, contact information, and exact dates other than years of treatment). Data were stored on institutional secure servers, accessible only to authorized study personnel under the institutional data-protection policy. Outputs generated by the language models were inspected by the study team before being shown to reviewers, and any incidental identifiable strings produced by the model were redacted before expert review. No additional compensation was provided. The manuscript and supplementary materials do not contain any images, screenshots, or quotations that could identify individual patients.</p></sec><sec id="s2-3"><title>Dataset Curation</title><sec id="s2-3-1"><title>SFT Dataset Construction</title><p>The SFT dataset requires training pairs of prompt inputs and ground truth outputs. For the input prompt, we collated 9 fields per patient, ranging from structured baseline data to unstructured textual descriptions. The structured data included female age, menstrual cycle, weight, BMI, anti-M&#x00FC;llerian hormone, follicle-stimulating hormone, and infertility years. The unstructured annotations included gynecology ultrasound reports and medical history, which typically document previous conditions such as surgical history, assisted reproduction history, and male semen analysis. Because the unstructured clinical narratives (gynecology ultrasound reports and medical history) were recorded in Chinese, all narratives were translated into English using ChatGPT-4o (OpenAI) under fixed prompts that instructed the model to retain medically important information, use professional or standard medical terminology, and translate accurately and concisely; the medical-history prompt additionally specified that the translation should be produced without speculation. For the ground truth output, we designed a double-layer structure to ensure explainability: (1) a multipart clinical reasoning (chain-of-thought [CoT]), and (2) the final diagnosis and treatment plan.</p><p>However, manually annotating thousands of clinical reasoning chains was neither time-permitted nor affordable. Therefore, we generated the CoT component using an ICL approach with a diverse case boutique prompt as a few-shot prompt. This ICL-based CoT generation approach was systematically validated in our previous work [<xref ref-type="bibr" rid="ref27">27</xref>], in which physicians confirmed that the generated reasoning chains were clinically accurate and consistent with expert-level diagnostic logic. The boutique set included 6 commonplace ART cases, with sample CoTs carefully curated by 2 expert-level physicians (these cases were excluded from our basic dataset). The resulting CoTs were structured into the following 4 aspects: diagnosis reasoning, assisted reproduction technology decision, ovarian stimulation protocol selection, and gonadotropin initiation dosing rationale. These 4 are relevant to part 2, diagnosis and treatment plan. In the second part, we set 5 final answer fields to mimic the physician&#x2019;s final decision-making annotation:</p><list list-type="bullet"><list-item><p>Diagnosis fields: infertility type (primary, secondary, or other) and initial differential diagnosis.</p></list-item><list-item><p>Treatment fields: ART strategy (11 subtypes), COS regimen (12 different protocols), and the gonadotropin starting dose.</p></list-item></list><p>For diagnosis, infertility type judgment and initial differential diagnosis were included. Each case was labeled with 1 of the following 3 infertility categories: primary infertility, secondary infertility, or other (unclear or multifactorial cases). The initial diagnosis mainly focuses on the female side&#x2019;s potential causes, and the male side is included if applicable. The ART strategy, the COS regimen, and the gonadotropin starting dose were used for the treatment plan. These three vital plans build the foundation of the following assisted reproduction and can be inferred through the comprehensive input information. The COS regimen also has 12 different protocols. In our training process, we separate the entire dataset into the training set (6559/8201, 80%), the validation set (821/8201, 10%), and the test set (821/8201, 10%)</p></sec><sec id="s2-3-2"><title>Pyramid Dataset Construction</title><sec id="s2-3-2-1"><title>Overview</title><p>Before any alignment-related processing, the full curated dataset of 8201 cases was partitioned into a training set of 7380 cases (90%, same as the SFT training set and validation set) and a test set of 821 cases (10%). The 821-case test set was used exclusively for the final evaluation of all 4 alignment strategies and was never accessed during SFT training, during pyramid dataset construction, or during any subsequent DPO or GRPO training step.</p><p>To further strengthen posttraining alignment and improve the model&#x2019;s discrimination on clinically ambiguous or low-frequency cases, we constructed a pyramid-style alignment dataset drawn entirely from the 7380-case training set. This dataset specifically addresses 2 limitations of the SFT baseline&#x2014;its bias toward dominant treatment categories and reduced robustness in rare or clinically complex scenarios. The pyramid consists of 3 layers (general enhancement, confusion enhancement, and human enhancement), described further in this study. After SFT was completed, we performed SFT inference on the 7380 training cases to obtain model-generated responses paired with the ground-truth annotations of the same training cases. Pyramid samples were drawn from these training-set inferences. SFT inference was also performed on the 821-case test set, but only for the purpose of reporting SFT evaluation metrics; test-set inferences were never included in the pyramid dataset and were never used to construct DPO preference pairs or GRPO reward signals. The full alignment dataset was then assembled in a top-down manner into top layer, middle layer, and bottom layer.</p></sec><sec id="s2-3-2-2"><title>Top Layer: Human Enhancement</title><p>This layer focuses on the most clinically challenging and sparsely represented cases. Two representative long-tail categories were selected: (1) IVF+ICSI, a hybrid of 2 ART generations that accounts for only 1.89% (155/8201) of SFT training data; and (2) short-protocol IVF, a complicated variant of standard IVF that is difficult even for specialists. A total of 50 cases (25 per category) were curated to maximize signal clarity and expose the model to high-value reasoning patterns rarely encountered in routine data.</p></sec><sec id="s2-3-2-3"><title>Middle Layer: Confusion Enhancement</title><p>This layer focuses on systematic confusion observed in the predictions of the initial SFT model. We evaluated the outputs across both ART strategy and COS regimen, using row-normalized confusion matrices (<xref ref-type="fig" rid="figure2">Figure 2</xref>) to identify high-confusion regions. A total of 628 samples were extracted from categories with individual error rates greater than 10%. Particularly, poor discrimination among nonantagonist COS regimens led to the inclusion of all other regimen types, allowing this layer to fully capture the blind spots of the model.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Cross-metric confusion analysis of the supervised fine-tuning baseline model on infertility decision tasks. (A) Row-normalized confusion matrix for controlled ovarian stimulation (COS) regimen selection across 12 protocol categories; (B) confusion matrix for ART strategy prediction across IVF, ICSI, and PGT. Color intensity indicating normalized frequency (e.g, blue=under-prediction, red=over-prediction). Each cell denotes the proportion of ground-truth cases assigned to a predicted category; diagonal cells represent correct predictions. These matrices reveal systematic confusion patterns, particularly the model&#x2019;s tendency to over-predict dominant categories such as the Antagonist regimen and IVF. High-confusion regions identified here informed the construction of the middle &#x201C;confusion enhancement&#x201D; layer of the pyramid alignment dataset.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97221_fig02.png"/></fig></sec><sec id="s2-3-2-4"><title>Bottom Layer: General Enhancement</title><p>To avoid overfitting toward rare or confusing samples, the base layer includes a balanced mixture that includes all groups of cases. This bottom layer contains 1000 samples. At least 700 samples with 1 incorrect field and 300 completely correct samples were included across the ART strategy and COS regimen tasks. This design is to provide stable generalization support and prevent distributional skew introduced by the upper layers.</p><p>The pyramid alignment dataset contains 1678 samples in total (1666 unique training-set cases, with a small number of cases appearing in more than 1 layer). This pyramid dataset was internally split 90%:10% into alignment-training and alignment-validation subsets, which were used to train and monitor the DPO and GRPO models. The 821-case held-out test set was not part of this internal split and was never used for pyramid construction, preference-pair construction, reward computation, model selection, or alignment monitoring. We verified that the intersection between held-out test-set case identifiers and pyramid-alignment case identifiers was zero. All final metrics reported in the Results section were computed exclusively on the 821-case held-out test set.</p></sec></sec></sec><sec id="s2-4"><title>Modeling</title><p>To investigate how different alignment paradigms influence clinical reasoning and decision-making, we compare representative strategies built on a shared backbone. Given the clinical nature of our task, we opted not to directly fine-tune a general-purpose pretrained model, such as Qwen-2.5 (Alibaba Cloud) or LLaMA-3 (Meta). Instead, we used OpenBioLLM-8b (Saama AI Labs) [<xref ref-type="bibr" rid="ref28">28</xref>], a domain-specific open-sourced model tailored for the medical field, which has demonstrated strong performance compared with other SOTA open-sourced models at the time of use. OpenBioLLM is built upon LLaMA-3 and has been trained on a range of medical knowledge sources. It has been adopted in several studies across clinical natural language processing and vision-language applications [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. We evaluate 4 alignment paradigms&#x2014;SFT, DPO, GRPO, and ICL&#x2014;chosen to represent supervised, preference-based, reinforcement-based, and prompt-based alignment strategies. These 4 paradigms jointly span the major alignment families used in contemporary LLM research, enabling us to systematically assess how algorithmic optimization interacts with clinical reasoning and trust.</p><p>In this study, we aim to obtain an LLM with specific capability in infertility diagnosis and treatment planning, accompanied by the reasoning text, or CoT on each aspect. We define infertility diagnosis and treatment planning as a structured, multioutput reasoning problem. Given a patient record &#x1D44B; = {&#x1D465;1, &#x1D465;2, &#x1D465;3 &#x2026; , &#x1D465;&#x1D45B;}, which integrates both structured baseline data and unstructured clinical narratives (eg, gynecology ultrasound reports and medical history), the model &#x1D453;<sub>&#x1D703;</sub> is required to generate 5 interdependent clinical decisions and their corresponding reasoning chains. The task can be expressed as learning a mapping function:</p><disp-formula id="E1"><label>(1)</label><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow></mml:msub><mml:mo>:</mml:mo><mml:mi>X</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>3</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>4</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>5</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where (&#x1D44C;1, &#x1D44C;2, &#x1D44C;3, &#x1D44C;4, and &#x1D44C;5) represent five structured outputs: (1) infertility type judgment, (2) initial diagnosis, (3) ART strategy, (4) COS regimen, (5) gonadotropin starting dose.</p><p>&#x1D436; denotes the CoT reasoning text that supports and explains each decision. The learning objective of the model is to maximize the joint likelihood of generating both the reasoning process and the final structured decisions:</p><disp-formula id="E2"><label>(2)</label><mml:math id="eqn2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi class="mathcal" mathvariant="script">L</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>3</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>4</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>5</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi class="mathcal" mathvariant="script">D</mml:mi></mml:mrow></mml:mrow></mml:munder><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>3</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>4</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mn>5</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mi>X</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>This provides a unified objective for multiple alignment paradigms. The model learns both to produce correct clinical outcomes and to generate transparent and clinically correct reasoning chains. We trained the models and evaluated their performance based on this objective function. A dual evaluation pipeline that includes both algorithmic metrics and human expert feedback was used. The whole process is presented in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p></sec><sec id="s2-5"><title>SFT</title><p>SFT is the foundation for all subsequent alignment strategies. A clinically coherent policy network pretrained with structured reasoning is formed. In this stage, the base model was trained using the training set. Each case includes a structured clinical description and an expert response for reasoning and diagnosis, as previously mentioned in the data section. Training was performed with low-rank adaptation (LoRA; learning rate=3e-5, batch size=4) for 10 epochs on a single A100 GPU. The resulting model serves as the reference policy for the subsequent stages. The number of training epochs was selected based on validation-set loss plateau. Training loss converged after epoch 7, and validation loss showed no further improvement after epoch 10; the checkpoint at epoch 10 was retained for downstream alignment experiments.</p></sec><sec id="s2-6"><title>DPO</title><p>This variant focuses on aligning model reasoning with expert preferences by directly optimizing the token-level log-likelihood contrast between preferred and dispreferred responses. Direct DPO is a critic-free reinforcement learning method that has recently emerged from the RLHF paradigm [<xref ref-type="bibr" rid="ref6">6</xref>]. It provides a relatively simple and stable alternative to reward-model&#x2013;based offline approaches for aligning language models with human preferences and has shown promising results in reducing undesirable behaviors in baseline models. Unlike traditional RLHF methods that rely on learning a separate reward model or value function, DPO directly optimizes the policy by contrasting preferred and dispreferred responses. Specifically, each training data point consists of a prompt <italic>x</italic>, a preferred response &#x1D466;<sub>&#x1D464;</sub>, and a less-preferred response &#x1D466;<sub>&#x03B9;</sub>. Given the prompt <italic>x</italic>, the DPO loss encourages the policy model to assign a higher likelihood to &#x1D466;<sub>&#x1D464;</sub> over &#x1D466;<sub>&#x03B9;</sub>, thereby aligning the model outputs more closely with human preferences. The DPO loss function is formulated as:</p><disp-formula id="E3"> <label>(3)</label><mml:math id="eqn3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mrow><mml:mi class="mathcal" mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">D</mml:mi><mml:mi mathvariant="normal">P</mml:mi><mml:mi mathvariant="normal">O</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">f</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>w</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:mi class="mathcal" mathvariant="script">D</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>w</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">f</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>w</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>&#x2212;</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">f</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>&#x1D70B;<sub>&#x1D703;</sub> is the current policy model, &#x03C3; is the sigmoid function, and &#x03B2; is a temperature parameter that controls the sharpness of the preference. This loss function assigns equal weight to all tokens in each response. By doing so, we aim to align the reasoning process and final diagnosis to the human level. Compared with the correct final answer, this setup helps the model to predict the correct outcome, as well as generating a clinically sound CoT. In the context of our study, where both the intermediate reasoning (eg, identifying relevant clinical clues) and the final decision (eg, treatment plan) are critical, such token-level uniform supervision helps ensure that the model learns to reflect expert-like logic across the entire response, rather than merely optimizing for the final output token. In our experiments, we use the SFT model described in SFT section as the reference policy. DPO was conducted on a single A100 GPU with <italic>&#x03B2;</italic>=.3, learning rate=3e-7, and batch size=8 for 1 epoch. Single-epoch DPO follows the typical protocol and is consistent with broader RLHF/DPO practice, in which multiepoch training over a fixed preference dataset typically leads to overoptimization on the preference signal and policy collapse away from the reference SFT model. To directly rule out undertraining as a confounder of the relative DPO underperformance, we additionally trained DPO under 2 longer configurations on the same preference dataset and identical hyperparameters: a 2-epoch configuration and a 10-epoch configuration with early stopping on validation loss. Field-level accuracy on the held-out test set degraded sharply rather than improved as training proceeded. Infertility-type accuracy fell from 92.2% (1 epoch) to 72% (2 epoch) to 53.6% (10 epoch with early stopping). COS regimen accuracy fell from 60.9% to 38% to 18%. ART strategy accuracy fell from 75.03% to 12% to 9.4%. Gonadotropin starting-dose mean absolute error (MAE) rose from 45.12 IU to 55.50 IU to 58.45 IU. This pattern is consistent with the well-documented DPO collapse phenomenon, and the 1-epoch checkpoint was retained as the reported configuration.</p></sec><sec id="s2-7"><title>GRPO</title><p>This variant leverages reinforcement-based relative optimization to enhance decision consistency while eliminating the computational overhead of value-function training. GRPO is an evolutionary variant of proximal policy optimization (PPO) [<xref ref-type="bibr" rid="ref31">31</xref>]. For PPO, its advantage is computed by applying generalized advantage estimate [<xref ref-type="bibr" rid="ref32">32</xref>], based on the rewards and a learned value function. Consequently, a value function needs to be trained alongside the policy model and using a per-token KL penalty from the reference model to mitigate overoptimization of the reward model. As the value function is involved in the training, it is typically another model of comparable size to the policy model, bringing a gargantuan computational burden. While for GRPO, it obviates an additional value function for advantage computation and instead uses the reward average of multiple outputs sampled from the same question as the baseline. More specifically, for each question <italic>q</italic>, GRPO samples a group of outputs {&#x1D45C;1, &#x1D45C;2, &#x2026; , &#x1D45C;&#x1D43A;} from the old policy &#x1D70B;<sub>&#x1D703;</sub> and then optimizes the policy model by maximizing the following objective:</p><disp-formula id="E4"><label>(4)</label><mml:math id="eqn4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable columnalign="left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">G</mml:mi><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">P</mml:mi><mml:mi mathvariant="normal">O</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi><mml:mo>&#x223C;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>Q</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msub><mml:mi>o</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:msubsup><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x223C;</mml:mo><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>G</mml:mi></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:munderover><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>o</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>o</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:munderover><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>o</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>q</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>o</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mo>&#x003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>o</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>q</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>o</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mo>&#x003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mover><mml:mi>A</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mtext>&#x00A0;</mml:mtext><mml:mi>clip</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>o</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>q</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>o</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mo>&#x003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>o</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>q</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>o</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mo>&#x003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B5;</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>&#x03B5;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>A</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">K</mml:mi><mml:mi mathvariant="normal">L</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">f</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where &#x03B5; and &#x03B2; are hyperparameters, and &#x1D434;&#x0302; <sub>{&#x1D456;,&#x1D461;}</sub> is the advantage calculated based on relative rewards of the outputs inside each group only, which will be detailed in the following subsections. GRPO offers a compelling solution for medical domains where training a value function is often impractical due to limited data, and where structured, interpretable supervision is essential for aligning model behavior with expert expectations.</p><p>Reward design follows the common practices [<xref ref-type="bibr" rid="ref8">8</xref>] by using the accuracy reward, aimed at achieving the &#x201C;algorithm alignment.&#x201D; Since we have 4 structured final answer fields, except for the initial diagnosis, which comprises long sentences, we used a combined final reward to aggregate 4 separate rewards:</p><disp-formula id="E5"><label>(5)</label><mml:math id="eqn5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">I</mml:mi><mml:mi mathvariant="normal">T</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">I</mml:mi><mml:mi mathvariant="normal">T</mml:mi></mml:mrow></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">O</mml:mi><mml:mi mathvariant="normal">S</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">O</mml:mi><mml:mi mathvariant="normal">S</mml:mi></mml:mrow></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">G</mml:mi><mml:mi mathvariant="normal">n</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">G</mml:mi><mml:mi mathvariant="normal">n</mml:mi></mml:mrow></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">T</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">T</mml:mi></mml:mrow></mml:mrow></mml:msubsup><mml:mo>.</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>The weight for each field was set to:</p><disp-formula id="E6"><label>(6)</label><mml:math id="eqn6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">I</mml:mi><mml:mi mathvariant="normal">T</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0.2</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">O</mml:mi><mml:mi mathvariant="normal">S</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0.3</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">G</mml:mi><mml:mi mathvariant="normal">n</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0.2</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">T</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0.3.</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>The weighting coefficients were chosen on clinical grounds rather than tuned by formal hyperparameter search. COS regimen and ART strategy directly determine the treatment course and downstream patient management and were therefore assigned the higher weight (0.3 each). Infertility type is largely derivable from structured baseline information and was assigned a lower weight (0.2). The gonadotropin starting dose is a continuous numerical decision and is additionally handled by a graded reward described below, so its weight in the combined accuracy reward was set to the same lower value (0.2). The 4 weights sum to 1.0. We did not perform a full reward-weight ablation at training time within the present study. Additionally, as the gonadotropin dose is a numerical answer, to evaluate the predicted gonadotropin starting dose, we designed a graded reward function that reflects real-world clinical practices. In dealing with clinical ovarian stimulation protocols, the gonadotropin dose is typically adjusted in increments of 25 IU. Thus, a deviation of 25 IU is clinically acceptable. Therefore, we assign:</p><disp-formula id="E7"><label>(7)</label><mml:math id="eqn7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">G</mml:mi><mml:mi mathvariant="normal">n</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left left" rowspacing="0.6em 0.6em 0.2em" columnspacing="1em" displaystyle="false"><mml:mtr><mml:mtd><mml:mn>1.0</mml:mn><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mtext>if&#x00A0;</mml:mtext><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>&#x2264;</mml:mo><mml:mn>25</mml:mn><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0.5</mml:mn><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mtext>if&#x00A0;</mml:mtext><mml:mn>25</mml:mn><mml:mo>&#x003C;</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>&#x2264;</mml:mo><mml:mn>50</mml:mn><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0.0</mml:mn><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mtext>otherwise.</mml:mtext></mml:mtd></mml:mtr></mml:mtable><mml:mo fence="true" stretchy="true" symmetric="true"/></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>This reward strategy helps the model to approximate the clinically preferred range, even if the value is not exactly matched. It reflects the tolerance that physicians often exhibit during dose adjustment in clinical practice.</p><p>GRPO training configuration: GRPO was implemented using the verl framework with the trainer configured for GRPO advantage estimation. The actor was initialized from the SFT checkpoint and fine-tuned with LoRA adapters (rank=64, &#x03B1;=32). Training used a learning rate of 3e-5 for 5 epochs (best at 1 epoch), with a training batch size of 16, and the PPO clipping ratio &#x03B5; at the verl default value of 0.2. The KL loss coefficient was 0.001 with the low-variance KL estimator, and the rollout group size was n=5 per prompt. Training was conducted on 1 node with 4 NVIDIA A100 GPUs.</p></sec><sec id="s2-8"><title>ICL: Guideline-Based Prompt Alignment</title><p>The primary purpose of using ICL is to enhance reasoning generalization through guideline-based prompts. ICL is a training-free post-alignment approach commonly applied to pretrained models. Conventional few-shot ICL relies on instance-level examples. Our variant repurposes clinical heuristic maps into textual guidance blocks. These guidelines are oriented from physicians&#x2019; decision flowcharts and are intended to translate structured expert reasoning paths (eg, stimulation protocol selection) into natural-language rules embedded in the prompt. This design transforms medical knowledge into interpretable prompt-based supervision, allowing the model to internalize domain heuristics without additional parameter updates. We proposed 2 guidelines on the ART strategy and COS regimen in this variant. The SFT model is retained as the core inference framework. This is mainly due to its domain-specific alignment, particularly its ability to maintain good output consistency and clinical reasoning. Two guideline blocks are inserted into the prompt for each case (refer to Apendix S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> for the full guideline blocks used in the ICL prompt). Additionally, we prepend the following instruction:</p><p><bold>&#x201C;DO NOT USE THE GUIDELINE UNLESS YOU ARE NOT SURE ABOUT YOUR ANSWER.&#x201D;</bold></p><p>This design helps the model to rely mainly on its internalized reasoning ability and consult the guideline only when uncertain. In this way, the instructional ICL serves as a lightweight and interpretable alternative to gradient-based alignment, effectively simulating real-world physician behavior, where decisions are primarily experience-driven, with reference to protocols only when ambiguity arises. We acknowledge that this gated prompt design relies on the language model&#x2019;s ability to assess its own epistemic uncertainty, a capability that contemporary LLMs are known to calibrate poorly. We intentionally retained this gating because it mirrors real-world physician behavior in which decisions are primarily experience-driven and external guidelines are consulted only when ambiguity arises. We did not evaluate alternative prompt configurations (eg, unconditional guideline conditioning, retrieval-augmented prompting, or chain-of-verification approaches) in this study. The ICL results reported here therefore characterize this specific gated prompt configuration as we implemented it and should not be read as a general claim about prompt-based alignment as a paradigm.</p></sec><sec id="s2-9"><title>Dual-Evaluation Protocol</title><p>We used a dual-layer evaluation framework to assess both quantitative task performance and qualitative clinical quality. This framework integrates (1) automatic field-level metrics and (2) blinded independent expert review. This design ensures comprehensive examination of both algorithmic accuracy and clinician-perceived interpretability.</p></sec><sec id="s2-10"><title>Automatic Evaluation</title><p>Model outputs were evaluated using standard metrics for structured clinical fields. Specifically, we compute accuracy and macro-<italic>F</italic><sub>1</sub> for categorical predictions, including infertility type, ART strategy, and COS regimen. Specifically, the macro-<italic>F</italic><sub>1</sub> is calculated with the following formula:</p><disp-formula id="E8"><label>(8)</label><mml:math id="eqn8"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">M</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">o</mml:mi></mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>K</mml:mi></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover><mml:mi>F</mml:mi><mml:msub><mml:mn>1</mml:mn><mml:mi>i</mml:mi></mml:msub></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Where &#x1D43E; denotes the total number of categories, &#x1D439;<sub>1&#x1D456;</sub> is the &#x1D439;<sub>1</sub> score of the &#x1D456; category.</p><p>Because the initial diagnosis field exhibits high variability and paraphrasing, exact word-by-word matching is infeasible and unreliable. Therefore, an auxiliary LLM was used as a semantic judge to determine whether the generated diagnosis semantically entailed the ground-truth diagnosis. We report the precision of containing at least one true diagnosis or all of the ground truth. For the numerical field of gonadotropin starting dose, prediction error was quantified using MAE. Together, these metrics assess the model&#x2019;s structured decision-making performance but do not capture the quality of its reasoning process.</p></sec><sec id="s2-11"><title>Doctor-in-the-Loop Evaluation</title><p>To evaluate clinical reasoning and decision-making quality beyond quantitative metrics, we conducted a domain-expert assessment. This evaluation addressed two core questions:</p><list list-type="order"><list-item><p>Is there a possible decoupling between &#x201C;algorithmic alignment&#x201D; and &#x201C;clinical alignment&#x201D;? Specifically, does aggressive outcome optimization come at the hidden cost of sacrificing reasoning quality and therapeutic feasibility?</p></list-item><list-item><p>How well do the aligned models simulate expert-level performance?</p></list-item></list><p>Experts rated each output on 4 clinical dimensions:</p><list list-type="bullet"><list-item><p>Clinical reasoning capability: whether the model is a medically sound thought process.</p></list-item><list-item><p>Diagnosis accuracy: whether the inferred infertility type and the Initial diagnosis are correct.</p></list-item><list-item><p>Treatment feasibility: whether the recommended treatment plan is appropriate and consistent.</p></list-item><list-item><p>Hallucination: whether the reasoning process contains irrelevant or wrong information.</p></list-item></list><p>These dimensions are a relatively comprehensive subjective evaluation. Each output used a 5-point Likert scale (1=poor, 5=excellent). The fourth dimension, Hallucination, was evaluated as a binary judgment (1=yes, 0=no) rather than on the Likert scale. We compare the outputs of the SFT model and the best-performing postalignment model for each evaluation case. Additionally, to further benchmark model performance against clinical standards, a third response from the ground truth was included, which is used to represent the expert-level response. The graders are then asked to select the overall best response. For the 3-way best-response task, all candidate responses were presented in a standardized response format containing a reasoning section and a final treatment-plan section. In the reference response, the diagnosis and treatment plan were the authentic physician-charted ground truth. The accompanying reasoning narrative was generated by ChatGPT-4o from that charted decision, using the same specialist-curated CoT prompt applied during SFT dataset construction, and was subsequently reviewed and edited by 2 reproductive medicine specialists. The clinical decision was therefore human-authored, whereas the reasoning narrative was model-drafted and physician-verified. To reduce source-recognition bias, response labels, model identifiers, and original chart-note markers were removed, and the 3 candidate responses were randomized before expert review. In a small number of cases reviewers indicated a tie between 2 or 3 responses (eg, both SFT and the charted plan rated as equivalently best); for 2-way ties, each tied option received 0.5; for 3-way ties, each option received 0.33. The per-rater fractional preferences were then averaged across the 2 reviewers (Y Long and TT) and the resulting proportions are reported.</p><p>All responses are blindly evaluated. Model identifiers and response labels were removed before review. However, because the 3-way comparison included a physician-certified reference response derived from charted care, residual source-recognition bias and partial unblinding cannot be fully excluded. Therefore, this comparison should be interpreted as a standardized best-response preference task rather than a fully clean blinded comparison between models and physician decision-making. The graders are unaware of the matching relationship to reduce bias. Each case is independently reviewed by 2 experienced physicians (Y Long and TT) in reproductive medicine. Considering the limited time available for physicians, we selected 100 representative cases as the evaluation set. The expert review should be interpreted as an exploratory clinical-quality assessment rather than a definitive clinical validation study.</p><p>Interrater reliability was quantified using 5 complementary metrics, because raw Cohen &#x03BA; is known to be unreliable when marginal rating distributions are highly skewed, a situation referred to as the &#x03BA; paradox. We report raw observed agreement (P_o), agreement within one Likert point (P_o&#x00B1;1), quadratic-weighted Cohen &#x03BA;, prevalence-adjusted bias-adjusted kappa (PABAK), and Gwet AC1 coefficient. The latter 2 metrics are designed to be robust to skewed marginal distributions. All metrics are reported per Likert dimension and pooled across the 3 dimensions evaluated by both reviewers.</p><p>To assess whether the custom graded gonadotropin starting-dose reward and the per-field correctness signals propagated into the subjective treatment-feasibility ratings, we conducted a post hoc correlation analysis between the per-case absolute gonadotropin dose error, per-case COS regimen correctness, per-case ART treatment correctness, and the per-case mean feasibility score (averaged across the two raters) for both SFT and GRPO outputs. Pearson and Spearman correlations and case-stratified mean comparisons (dose-correct vs dose-incorrect, and by tertile of dose error) are reported in the Results section.</p><p>All outputs previously flagged as hallucinated were subsequently reviewed and categorized into low-severity, moderate-severity, or potentially unsafe hallucinations according to whether the unsupported content could alter diagnosis, ART strategy, COS regimen, gonadotropin dosing, or patient safety. All hallucination-positive rater&#x2013;output judgments were subsequently reviewed by the same 2 reproductive medicine specialists who performed the original expert evaluation. The specialists remained blinded to model identity, and disagreements were resolved by consensus.</p></sec><sec id="s2-12"><title>Statistical Analysis</title><p>For automatic evaluation, accuracy, macro-<italic>F</italic><sub>1</sub>, and MAE were reported as point estimates across all structured decision fields, with 95% CIs computed via nonparametric bootstrap resampling (n=2000 iterations stratified by case). The point estimates themselves were not subjected to inferential significance testing because they are deterministic functions of the model outputs on a fixed held-out test set. To assess whether differences in per-case classification correctness between models were unlikely to arise from sampling variation, we additionally performed pairwise comparisons on categorical fields (infertility type, ART strategy, and COS regimen) using the exact McNemar test on the per-case correctness indicator. Holm-Bonferroni correction was applied across the 9 pairwise model comparisons. Both raw and Holm-adjusted <italic>P</italic> values are reported.</p><p>For the blinded independent expert review, three clinical dimensions (clinical reasoning capability, diagnostic accuracy, and treatment feasibility) were scored on a 5-point Likert scale. Hallucination was evaluated separately as a binary per-output judgment and summarized as a rate. Likert ratings are ordinal; we therefore report medians and IQRs as the primary descriptive statistics, with means and SDs reported alongside for comparability with previous work. Pairwise comparisons of expert-rated dimensions between SFT and GRPO used Wilcoxon signed-rank tests on the per-case mean of the 2 raters, with Holm-Bonferroni correction across the 3 Likert dimensions. Effect sizes are reported as paired Cohen <italic>d</italic>, with the limitations of treating ordinal Likert data as continuous explicitly noted in the Limitations section. Interrater reliability was quantified using multiple complementary metrics, including raw observed agreement, within-one-point agreement, quadratic-weighted Cohen &#x03BA;, PABAK, and Gwet AC1 coefficient, because raw Cohen &#x03BA; is uninformative when marginal rating distributions are highly skewed. For clarity, differences between model accuracies are reported as absolute percentage-point differences, with relative percent changes provided in parentheses when useful.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Automatic Metrics Performance</title><p>Reinforcement-based alignment achieves superior algorithmic precision across structured decision layers. GRPO establishes a higher automatic field-level performance over SFT (baseline model) in objective outcome metrics. <xref ref-type="table" rid="table1">Table 1</xref> summarizes the field-level performance of the 4 models. GRPO (the SFT model trained by GRPO) achieves relatively the best overall results in the fields of Infertility type, ART strategy, and COS regimen. Individually, GRPO has the highest accuracy of 92.57% of infertility type judgment, 76.49% of ART strategy selection, and 89.4% of initial diagnosis partial match, and 20.32% of exact match. Compared with the base SFT model, GRPO improves ART strategy selection accuracy by 2.68% points, though this did not survive correction for multiple comparisons (McNemar p_raw=0.020, p_Holm=0.118). As for the unstructured Initial diagnosis task, GRPO also demonstrates clear enhancement, with a 1.95 percentage points higher partial match rate and a 4.24% points increase in exact matches. Overall, GRPO achieves the highest average accuracy (77.14%) and macro-<italic>F</italic><sub>1</sub> score (50.64%). For other models, DPO shows moderate gains in ART strategy choice (accuracy+1.22% vs SFT), which indicates the effectiveness of the preference alignment training strategy. However, its instability is reflected in the initial diagnosis task. DPO showed a substantial decline in initial-diagnosis performance, with partial match decreasing by 46.16% points relative to SFT (87.45% to 41.29%) and exact match decreasing by 11.45% points (16.08%-4.63%). While for the last model, although built on the same SFT backbone, ICL failed in nearly all metrics, with only a marginal improvement in ART strategy selection (accuracy+0.95%). Notably, all models struggle with the COS regimen prediction. This may be due to the possible inherently ambiguous category boundaries. As the overall supreme model in the other field, GRPO records a slight decrease in accuracy (&#x2212;1.34 percentage points vs SFT) and <italic>F</italic><sub>1</sub> (&#x2212;0.88 vs SFT), while DPO and ICL exhibit a larger accuracy drop (&#x2212;2.80%, &#x2212;4.26% vs SFT), despite a minor <italic>F</italic><sub>1</sub> improvement by DPO (+0.56 vs SFT).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Field-level performance comparison of SFT<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>, ICL<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>, DPO<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>, and GRPO<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> models on infertility-related decision tasks. Results are reported for 5 evaluation fields: infertility type classification, ART<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup> strategy selection, COS<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup> regimen prediction, initial diagnosis (partial and exact match), and gonadotropin starting dose (IU). Overall, GRPO demonstrates the highest accuracy and F1 score.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="2">Infertility type</td><td align="left" valign="bottom" colspan="2">ART strategy</td><td align="left" valign="bottom" colspan="2">COS regimen</td><td align="left" valign="bottom" colspan="2">Average</td><td align="left" valign="bottom" colspan="2">Initial diagnosis</td><td align="left" valign="bottom">Gonadotropin starting dose (IU)</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Macro <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Macro <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Macro <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom">Macro <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Partial</td><td align="left" valign="bottom">Exact</td><td align="left" valign="bottom">MAE<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup> (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">SFT</td><td align="left" valign="top">92.57<break/>(90.7&#x2010;94.4)</td><td align="left" valign="top">92.18<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">73.81 (70.8&#x2010;76.7)</td><td align="left" valign="top">49.18</td><td align="left" valign="top">63.70 (60.4&#x2010;66.8)<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">7.75</td><td align="left" valign="top">76.69</td><td align="left" valign="top">49.70</td><td align="left" valign="top">87.45</td><td align="left" valign="top">16.08</td><td align="left" valign="top">43.88 (40.7&#x2010;47.0)<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td></tr><tr><td align="left" valign="top">ICL</td><td align="left" valign="top">90.83 (88.4&#x2010;92.4)</td><td align="left" valign="top">89.88</td><td align="left" valign="top">74.76 (72.1&#x2010;77.8)</td><td align="left" valign="top">33.31</td><td align="left" valign="top">59.44 (56.0&#x2010;62.6)</td><td align="left" valign="top">5.89</td><td align="left" valign="top">75.01</td><td align="left" valign="top">43.03</td><td align="left" valign="top">87.21</td><td align="left" valign="top">16.09</td><td align="left" valign="top">48.38 (45.2&#x2010;51.6)</td></tr><tr><td align="left" valign="top">DPO</td><td align="left" valign="top">92.20 (90.3&#x2010;94.0)</td><td align="left" valign="top">91.33</td><td align="left" valign="top">75.03 (72.4&#x2010;78.3)</td><td align="left" valign="top">46.62</td><td align="left" valign="top">60.90 (57.6&#x2010;64.1)</td><td align="left" valign="top">8.31<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">76.04</td><td align="left" valign="top">48.75</td><td align="left" valign="top">41.29</td><td align="left" valign="top">4.63</td><td align="left" valign="top">45.12 (42.0&#x2010;51.6)</td></tr><tr><td align="left" valign="top">GRPO</td><td align="left" valign="top">92.57 (90.6&#x2010;94.3)<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">92.05</td><td align="left" valign="top">76.49 (73.9&#x2010;79.7)<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">52.99</td><td align="left" valign="top">62.36 (59.0&#x2010;65.4)</td><td align="left" valign="top">6.87</td><td align="left" valign="top">77.14<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">50.64<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">89.40<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">20.32<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">44.94 (41.7&#x2010;48.2)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>SFT: supervised fine-tuning.</p></fn><fn id="table1fn2"><p><sup>b</sup>ICL: in-context learning.</p></fn><fn id="table1fn3"><p><sup>c</sup>DPO: direct preference optimization.</p></fn><fn id="table1fn4"><p><sup>d</sup>GRPO: group relative policy optimization.</p></fn><fn id="table1fn5"><p><sup>e</sup>ART: assisted reproductive technology.</p></fn><fn id="table1fn6"><p><sup>f</sup>COS: controlled ovarian stimulation.</p></fn><fn id="table1fn7"><p><sup>g</sup>MAE: mean absolute error.</p></fn><fn id="table1fn8"><p><sup>h</sup>highlights the best performance in each metric.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Doctor-in-the-Loop Evaluation</title><p>Our results suggest that algorithmic gains may not automatically lead to clinical trust. In contrast, the evaluation results show that the reasoning clarity and regimen feasibility of the baseline output are directionally higher rated by practitioners compared to others. We used human feedback from two reproductive experts to evaluate the models&#x2019; clinical performance in real-world workflow. We selected the baseline model (SFT) and the posttraining model (GRPO), which achieved the highest automatic scores. Each model&#x2019;s output was scored on 4 primary dimensions: diagnosis accuracy, clinical reasoning capability (reasoning), treatment plan clinical feasibility (feasibility), and hallucination. The rating distributions across the 2 reviewers, the 3 Likert dimensions, and the 2 models were strongly concentrated at the upper end of the scale (62.3% of ratings were 4, 23.8% were 5, 13.8% were 3, and fewer than 0.3% were 1 or 2). Under this regime raw Cohen &#x03BA; is uninformative because chance agreement is already very high. The 2 reviewers gave the same Likert score on 41% of pooled ratings and agreed within one Likert point on 93.7% of pooled ratings. Quadratic-weighted Cohen &#x03BA; was 0.05 pooled (range 0.03 to 0.06 across dimensions). PABAK was 0.26 pooled (accuracy=0.30, reasoning=0.28, and feasibility=0.21). Gwet AC1 was 0.32 pooled (accuracy=0.35, reasoning=0.33, and feasibility=0.28). Both kappa-paradox-robust metrics fall in the fair-to-moderate range on the Landis and Koch scale, consistent with the high within-one-point agreement.</p><p>The results of this expert scoring (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>) revealed a notable discrepancy with the findings from the automated metrics. Median scores were 4.00 (IQR 4.00&#x2010;4.50) for SFT versus 4.00 (IQR 3.50&#x2010;4.50) for GRPO on diagnostic accuracy, 4.00 (IQR 4.00&#x2010;4.50) for SFT versus 4.00 (IQR 4.00&#x2010;4.50) for GRPO on reasoning capability, and 4.50 (IQR 4.00&#x2010;4.50) for SFT versus 4.00 (IQR 4.00&#x2010;4.50) for GRPO on treatment feasibility. The corresponding means were 4.04 (SD 0.42) vs 3.99 (SD 0.45), 4.17 (SD 0.42) vs 4.08 (SD 0.42), and 4.21 (SD 0.39) vs 4.09 (SD 0.42). Pairwise comparisons used Wilcoxon signed-rank tests on the per-case mean of the 2 raters with Holm-Bonferroni correction across the 3 dimensions; Holm-adjusted <italic>P</italic> values were .21 for diagnostic accuracy, .07 for reasoning capability, and .06 for treatment feasibility. Nonparametric (rank-biserial) effect sizes were 0.21, 0.30, and 0.33 respectively (small to small-to-moderate by Cohen conventions); the corresponding paired Cohen <italic>d</italic> values, presented for comparison with previous work and under the explicit assumption that Likert data are treated as continuous, were +0.13, +0.22, and +0.25.</p><p>Despite GRPO showed stronger overall automatic performance than SFT across several fields, the physician ratings suggested an opposite directional trend. The SFT model received directionally higher ratings for reasoning capability and treatment feasibility (mean SFT 4.17, SD 0.42; mean GRPO 4.08, SD 0.42; Cohen <italic>d</italic>=0.22, mean difference 95% CI 0.005-0.170; <italic>P</italic>=.03, adjusted <italic>P</italic>=.07) and clinical feasibility (mean SFT 4.21, SD 0.39; mean GRPO 4.09, SD 0.42; Cohen <italic>d</italic>=0.25, mean difference 95% CI 0.025-0.210; <italic>P</italic>=.02, adjusted <italic>P</italic>=.06). There were no significant differences observed in diagnosis accuracy between the 2 models. For the additional dimension, hallucination, GRPO was rated to have a lower rate (15%) compared with SFT (18.5%). Among outputs flagged as hallucinated, most hallucinations were low-severity extraneous statements rather than safety-critical treatment errors. Specifically, 82.1% (55/67) were classified as low severity, 3% (2/67) as moderate severity, and 14.9% (10/67) as potentially unsafe. Low severity denotes a benign, clinically inert fabrication that does not alter the infertility diagnosis, ART indication, stimulation protocol, or dosing (eg, an invented &#x201C;breast cyst&#x201D;). Moderate severity denotes a false but noncommunicable condition or history that could prompt unnecessary work-up, referral, or patient anxiety without directly dictating harmful treatment (eg, recasting a single biochemical pregnancy as &#x201C;recurrent pregnancy loss&#x201D;). Potentially unsafe denotes a fabrication that could cause real harm if acted upon, which may trigger unnecessary treatment, partner notification, biosafety measures, stigma, or delay of time-sensitive care. This severity distribution explains why hallucination flags could coexist with high global Likert ratings: the Likert dimensions captured overall reasoning coherence and treatment feasibility, whereas the binary hallucination item captured any unsupported content, including low-impact extraneous information.</p><p>Our results suggest a possible trade-off. The optimization for outcome token-level precision may be the result of sacrificing process transparency and output coherence. This indicates an &#x201C;alignment paradox&#x201D; phenomenon. The pursuit of higher benchmark scores may lead to a negative result of a mismatch between model performance and physician expectations.</p><p>To evaluate whether the custom graded gonadotropin dose reward and the per-field correctness signals propagated into the subjective treatment-feasibility ratings, we computed the per-case association between automatic correctness and physician-rated feasibility on the same 100-case evaluation sample. The absolute gonadotropin starting-dose error was not meaningfully correlated with mean physician-rated treatment feasibility for either model (SFT: Pearson <italic>r</italic>=&#x2212;0.11, <italic>P</italic>=.30; Spearman &#x03C1;=&#x2212;0.09, <italic>P</italic>=.40; GRPO: Pearson r=+0.09, <italic>P</italic>=.39; Spearman &#x03C1;=+0.08, <italic>P</italic>=.41). Stratified comparison gave the same conclusion: mean feasibility on cases for which gonadotropin dose was exactly correct versus cases for which it was incorrect was 4.23 versus 4.19 for SFT and 4.07 versus 4.10 for GRPO (absolute differences &#x003C;0.05 on a 5-point scale, with nonmonotonic patterns across error tertiles). COS regimen correctness and ART treatment correctness were similarly weak drivers of feasibility scores (absolute differences &#x2264;0.07 between correct and incorrect subgroups for both models). These results indicate that the 2 reproductive medicine specialists rated treatment feasibility based on the overall coherence and clinical sensibility of the treatment plan as a whole rather than on the correctness of any single structured decision field. The custom graded gonadotropin dose reward used during training therefore did not propagate a confounding signal into the physician-rated feasibility comparison.</p></sec><sec id="s3-3"><title>Winning Rate</title><p>In addition to the process-level ratings, we included an overall best-response question, which asks the experts to blindly identify the best response from three responses, including 2 scored responses by models and a third response representing the physician-certified reference response. As illustrated in subpart (B) in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>, physicians selected the output by the SFT model in more than half of the cases (51.2%, 95% CI 44.2%-58.1%), higher than the GRPO outputs (26.2%, 95% CI 20.1%-32.3%) and the original charted plans (22.6%, 95% CI 16.9%-28.5%) in this structured preference task. This result should be interpreted as a preference within the review format rather than as evidence that model-generated decisions are clinically superior to physician decision-making. When the 3-way preference is reduced to pairwise comparisons, SFT was selected over GRPO in 66.2% of head-to-head matchups (Holm-adjusted <italic>P</italic>&#x003C;.001) and over the original physician-charted plan in 69.3% (Holm-adjusted <italic>P</italic>&#x003C;.001), while GRPO did not differ significantly from the charted plan (53.6%, Holm-adjusted <italic>P</italic>=.48). These results should be interpreted cautiously, given the substantial between-rater variability: one evaluator showed a strong SFT preference (77.5% SFT over GRPO) while the other showed a more balanced distribution (54% vs 46%). This preference for model outputs over charted plans may have been partially influenced by the well-structured and completeness of model responses, which charted plans may not contain. The SFT-versus-GRPO contrast within this comparison is less subject to this consideration because both model outputs include reasoning chains generated directly by models. Specifically, it shows the same trend as the results shown in human evaluations. However, physicians are inclined to the baseline response for its reasoning clarity and decision coherence. In essence, interpretability and accuracy may be decoupled through optimization, which elevated benchmark scores but eroded the coherence that is essential for clinical trust.</p><p>Collectively, these three layers of evaluation above interpret the possible nature of the &#x201C;alignment paradox.&#x201D; Automatic metrics illustrate the effectiveness of reinforcement learning&#x2013;based alignment in maximizing token-level outcome precision. At the same time, physicians&#x2019; ratings suggest that the simplest baseline results in the highest level of reasoning accuracy. Our blinded winning rates comparison indicates that although both paradigms have better consistency than humans, the mechanisms used to achieve this are fundamentally different. GRPO&#x2019;s stronger automatic performance and directionally lower clinician ratings suggest that the current optimization goals divide &#x201C;accuracy&#x201D; from &#x201C;logic.&#x201D; As a result, the consistency between the process and the final answers may contribute to clinical trust.</p></sec><sec id="s3-4"><title>Subgroup Analysis</title><p>In high-stakes fields such as clinical question answering, performance consistency across clinical categories is as important as overall performance. Thus, how these methods influence the model output under various clinical contexts is critical. We evaluated model behavior across major ART categories (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). The current ART is evaluated in three main categories: IVF, ICSI, and PGT. For the PGT category, GRPO achieved a higher <italic>F</italic><sub>1</sub> score than SFT (SFT=0.921; GRPO=0.934). For IVF, GRPO and DPO yielded <italic>F</italic><sub>1</sub> scores comparable to and slightly higher than SFT (SFT=0.924; GRPO=0.926; DPO=0.925). By contrast, in the ICSI category, GRPO and DPO yielded lower <italic>F</italic><sub>1</sub> scores than SFT (SFT=0.721; GRPO=0.687; DPO=0.709).</p><p>To examine the origins of these trends, we further conducted a fine-grained analysis of more detailed ART subtypes (<xref ref-type="table" rid="table2">Table 2</xref>). We used the baseline SFT and GRPO for comparison. The differences between the 2 models are illustrated in <xref ref-type="fig" rid="figure3">Figure 3</xref>. GRPO has poorer performance across all ICSI-related subgroups. Among subgroups with at least 10 cases, the largest <italic>F</italic><sub>1</sub> decrease occurred in IVF+ ICSI (from 6.2% to 0%, a decrease of 6.2% points; n=16), followed by standard ICSI (a decrease of 5.93% points). GRPO produces better gains in complex PGT subtypes, for example, PGT-M (+20.9%) and PGT-SR (+1.7%). A similar pattern was noticed for IVF subtypes. GRPO performs better in Short Protocol IVF (+6.5%) and has a moderate gain in standard IVF (+2.8%). However, we would like to acknowledge that the lower performance in ICSI with frozen sperm or donor sperm should be interpreted carefully, as the sample sizes were limited (n=3 and n=1, respectively). Subtypes with n &#x003C;10 were excluded from comparative inference and are reported descriptively only.</p><p>Our findings map a detailed influence pattern of alignment in infertility medicine. While GRPO performs well for rare and complex problems (eg, PGT) through targeted reinforcement, its capability for intermediate categories like ICSI is very limited. Thus, to design trustworthy medical LLMs, it is necessary to move beyond simple data balancing to implementing adaptive fair rewards. The optimization of rare cases should not lead to decreased performance of the decision boundaries that are essential for routine care.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>ART<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> subtype performance. ART subtype-level performance of the SFT<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> baseline and the GRPO<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>-aligned model on the held-out test set. Per-subtype precision, recall, and <italic>F</italic><sub>1</sub>-score (percentages) are reported alongside test-set sample size. Subtypes with n&#x003C;10 are excluded and should be interpreted descriptively only.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">ART</td><td align="left" valign="bottom">N</td><td align="left" valign="bottom" colspan="3">SFT</td><td align="left" valign="bottom" colspan="3">GRPO</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom">Precision, % (95% CI)</td><td align="left" valign="bottom">Recall, % (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Precision, % (95% CI)</td><td align="left" valign="bottom">Recall, % (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">IVF<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">483</td><td align="left" valign="top">79.9 (76.3&#x2010;83.1)</td><td align="left" valign="top">86.5 (83.2&#x2010;89.3)</td><td align="left" valign="top">83.1</td><td align="left" valign="top">80.7 (77.2&#x2010;83.8)</td><td align="left" valign="top">91.7 (88.9&#x2010;93.9)</td><td align="left" valign="top">85.9</td></tr><tr><td align="left" valign="top">Standard ICSI<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top">96</td><td align="left" valign="top">57.0 (47.5&#x2010;66.0)</td><td align="left" valign="top">63.5 (53.6&#x2010;72.5)</td><td align="left" valign="top">60.1</td><td align="left" valign="top">54.2 (44.2&#x2010;63.8)</td><td align="left" valign="top">54.2 (44.2&#x2010;63.8)</td><td align="left" valign="top">54.2</td></tr><tr><td align="left" valign="top">Short Protocol IVF</td><td align="left" valign="top">76</td><td align="left" valign="top">30.0 (16.7&#x2010;47.9)</td><td align="left" valign="top">11.8 (6.4&#x2010;21.0)</td><td align="left" valign="top">17.0</td><td align="left" valign="top">46.2 (28.8&#x2010;64.5)</td><td align="left" valign="top">15.8 (9.3&#x2010;25.6)</td><td align="left" valign="top">23.5</td></tr><tr><td align="left" valign="top">PGT-SR<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="left" valign="top">49</td><td align="left" valign="top">83.6 (71.7&#x2010;91.1)</td><td align="left" valign="top">93.9 (83.5&#x2010;97.9)</td><td align="left" valign="top">88.5</td><td align="left" valign="top">85.2 (73.4&#x2010;92.3)</td><td align="left" valign="top">95.8 (86.0&#x2010;98.8)</td><td align="left" valign="top">90.2</td></tr><tr><td align="left" valign="top">TESA<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup>+ICSI</td><td align="left" valign="top">33</td><td align="left" valign="top">86.2 (69.4&#x2010;94.5)</td><td align="left" valign="top">75.8 (59.0&#x2010;87.2)</td><td align="left" valign="top">80.6</td><td align="left" valign="top">85.7 (68.5&#x2010;94.3)</td><td align="left" valign="top">72.7 (55.8&#x2010;84.9)</td><td align="left" valign="top">78.7</td></tr><tr><td align="left" valign="top">PGT-A<sup><xref ref-type="table-fn" rid="table2fn8">h</xref></sup></td><td align="left" valign="top">27</td><td align="left" valign="top">95.5 (78.2&#x2010;99.2)</td><td align="left" valign="top">77.8 (59.2&#x2010;89.4)</td><td align="left" valign="top">85.7</td><td align="left" valign="top">88.5 (71.0&#x2010;96.0)</td><td align="left" valign="top">82.1 (64.4&#x2010;92.1)</td><td align="left" valign="top">85.2</td></tr><tr><td align="left" valign="top">IVF with donor sperm</td><td align="left" valign="top">21</td><td align="left" valign="top">88.9 (67.2&#x2010;96.9)</td><td align="left" valign="top">76.2 (54.9&#x2010;89.4)</td><td align="left" valign="top">82.1</td><td align="left" valign="top">94.1 (73.0&#x2010;99.0)</td><td align="left" valign="top">76.2 (54.9&#x2010;89.4)</td><td align="left" valign="top">84.2</td></tr><tr><td align="left" valign="top">PGT-M<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td><td align="left" valign="top">16</td><td align="left" valign="top">77.8 (45.3&#x2010;93.7)</td><td align="left" valign="top">43.8 (23.1&#x2010;66.8)</td><td align="left" valign="top">56.0</td><td align="left" valign="top">100.0 (72.2&#x2010;100.0)</td><td align="left" valign="top">62.5 (38.6&#x2010;81.5)</td><td align="left" valign="top">76.9</td></tr><tr><td align="left" valign="top">IVF+ICSI</td><td align="left" valign="top">16</td><td align="left" valign="top">6.2 (1.1&#x2010;28.3)</td><td align="left" valign="top">6.2 (1.1&#x2010;28.3)</td><td align="left" valign="top">6.2</td><td align="left" valign="top">0.0 (0.0&#x2010;29.9)</td><td align="left" valign="top">0.0 (0.0&#x2010;19.4)</td><td align="left" valign="top">0.0</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>ART: assisted reproductive technology.</p></fn><fn id="table2fn2"><p><sup>b</sup>SFT: supervised fine-tuning.</p></fn><fn id="table2fn3"><p><sup>c</sup>GRPO: group relative policy optimization.</p></fn><fn id="table2fn4"><p><sup>d</sup>IVF: in vitro fertilization.</p></fn><fn id="table2fn5"><p><sup>e</sup>ICSI: intracytoplasmic sperm injection.</p></fn><fn id="table2fn6"><p><sup>f</sup>PGT-SR: preimplantation genetic testing for structural rearrangements.</p></fn><fn id="table2fn7"><p><sup>g</sup>TESA: testicular sperm aspiration.</p></fn><fn id="table2fn8"><p><sup>h</sup>PGT-A: preimplantation genetic testing for aneuploidy.</p></fn><fn id="table2fn9"><p><sup>i</sup>PGT-M: preimplantation genetic testing for monogenic disorders.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Differences between group relative policy optimization (GRPO) and supervised fine-tuning (SFT) in subgroups. Field-level differences between GRPO and SFT across detailed assisted reproductive technology (ART) subtypes in the held-out test set. (A) Per-subtype <italic>F</italic><sub>1</sub> difference (GRPO&#x2212;SFT, percentage points) across ART subtypes. Positive values indicate improvement under GRPO. Subgroups with fewer than 10 cases are excluded and reported descriptively only. (B) Radar plot comparing subtype-level <italic>F</italic><sub>1</sub> scores between SFT and GRPO.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97221_fig03.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>Our principal finding is that, in this single-center retrospective evaluation of mainstream alignment paradigms for medical LLMs in ART decision support, automatic field-level accuracy and physician-rated clinical quality suggested a possible decoupling. GRPO consistently produced the highest automatic accuracy across structured decision fields; yet, 2 blinded independent expert reviewers (Y Long and TT) had marginally higher ratings of the conservative SFT baseline on reasoning capability and treatment feasibility, with no significant difference in diagnostic accuracy. We refer to this decoupling as the alignment paradox: posttraining that improves token-level outcome metrics may simultaneously degrade the process-level reasoning that clinicians use to decide whether a model output is trustworthy. The alignment paradox interpretation does not rest on any single source of evidence. Three measurements with different statistical assumptions and different vulnerabilities converge on the same direction. First, 2 blinded reviewers (Y Long and TT) rated the SFT baseline higher than GRPO on reasoning capability and treatment feasibility, with small paired effect sizes (Cohen <italic>d</italic>=+0.22 and+0.25) and Holm-adjusted Wilcoxon <italic>P</italic> values of .07 and .06. Second, in the discrete three-way best-response preference (SFT 51.2% vs GRPO 26.2% vs original charted plan 22.6%), reviewers selected SFT substantially more often than GRPO; this comparison is a discrete three-category choice categorical choice and does not depend on the absolute calibration of Likert ratings. Although all 3 responses included explicit reasoning, the reference reasoning was itself model-drafted and physician-verified (produced by the same pipeline as the SFT training targets), so all three candidates shared a similar reasoning style; residual presentation-style differences and the greater completeness of the model outputs may nonetheless have influenced the preference. Therefore, the 3-way best-response finding should be interpreted cautiously and should not be taken as evidence of model superiority over physician decision-making. Third, the field-level subgroup pattern in which GRPO reduced <italic>F</italic><sub>1</sub> in ICSI cases is derived entirely from automatic metrics and is independent of any human rating. The convergence across these three independent measurements is the basis for the alignment paradox interpretation, and we discuss the interrater reliability findings that motivate this multisource framing in the Limitations section. The physician evaluation focused on SFT and GRPO because these two models represented the most clinically informative contrast in our experimental setting: SFT served as the supervised reasoning baseline, whereas GRPO achieved the strongest automatic metric performance. A full physician evaluation of all 4 paradigms would further strengthen the clinical interpretation of these findings and should be prioritized in future validation studies.</p><p>In addition, in the blinded preference settings of this study, expert reviewers tended to select LLM-generated responses more frequently than the original charted treatment plans. This finding should not be interpreted as evidence that model-generated decisions are clinically superior to real-world physician decision-making. Rather, it likely reflects the value of clear, standardized, and explicitly reasoned output presentation. Original clinical plans in electronic health records are often concise and decision-oriented, whereas model outputs are inherently more verbose and structured. Therefore, comparisons between model outputs and charted treatment plans may be affected by format asymmetry, while the comparison between SFT and GRPO is less affected because both models generated responses in the same format.</p></sec><sec id="s4-2"><title>Interpretation of Findings</title><p>This reverse pattern is consistent with concerns raised in the broader RLHF and process-supervision literature, where reward overoptimization is known to compress reasoning diversity and amplify confidently stated yet weakly justified outputs. Because our reward function scored only the correctness of final structured fields, GRPO had no explicit incentive to preserve the multistep clinical reasoning that humans evaluated. In complex, multidimensional decision systems, such as infertility treatment, many components of the reasoning chain are interdependent and difficult to evaluate using isolated final-field accuracy alone. Thus, an answer may be statistically correct at the field level while remaining less clinically interpretable, less feasible, or less trustworthy from the perspective of a physician reviewer.</p><p>The ICL variant evaluated here used a gated &#x201C;use-the-guideline-when-uncertain&#x201D; prompt that relies on the model&#x2019;s self-assessed epistemic uncertainty. Because contemporary LLMs are known to calibrate this capability poorly, the reported ICL performance reflects this specific prompt configuration rather than a general property of guideline-based prompting. Alternative prompting strategies should be evaluated in future work before drawing broader conclusions about the ICL paradigm in clinical reasoning tasks.</p><p>The hallucination analysis also clarifies an apparent tension between the binary hallucination rates and the high Likert-score distribution. Hallucination was operationalized as a sensitive binary safety screen and therefore included low-severity extraneous or unsupported statements, not only clinically dangerous recommendations. Most flagged hallucinations did not alter the final treatment plan, which likely explains why they were not consistently reflected in lower reasoning or feasibility scores. Nevertheless, unsupported content remains clinically undesirable because it may reduce trust, increase cognitive burden, or become consequential in different clinical contexts. Future evaluation rubrics should therefore separate low-impact unsupported statements from safety-critical hallucinations and should explicitly penalize errors that could change diagnosis, ART strategy, COS regimen, or gonadotropin dosing.</p><p>The subgroup analyses suggest a hypothesis-generating pattern: GRPO improved <italic>F</italic><sub>1</sub> substantially in rare and complex categories such as PGT-M and PGT-SR, but reduced <italic>F</italic><sub>1</sub> in ICSI-related scenarios. We attribute this to a data structuring choice in our dataset, male-factor diagnostic parameters such as semen analysis were retained only in unstructured fields, while structured baseline data were predominantly female-factor, which limited the reward signal available for ICSI-relevant decisions. In other words, even when reinforcement learning works as designed, it can amplify rather than correct upstream representational asymmetries in the training data. A complementary interpretation is that ICSI-related decisions may occupy a more ambiguous boundary region between common IVF pathways and more specialized treatment categories. In such settings, reinforcement learning optimized for final-field correctness may favor stable, high-frequency decision patterns rather than preserving sensitivity to individualized edge cases. We did not directly quantify decision-boundary uncertainty, subgroup calibration, or confidence around these overlapping categories; therefore, this interpretation should be regarded as hypothesis-generating. Importantly, this pattern should not be interpreted as evidence of an intrinsic limitation of GRPO or reinforcement learning itself. The observed decline in ICSI-related cases may reflect the interaction between single-center case mix, institution-specific ART decision preferences, male-factor information being retained primarily in unstructured narratives, and the specific static reward configuration used in this study. External validation is required to distinguish these explanations. Future studies should examine whether boundary-focused evaluation, subgroup-sensitive reward scaling, or uncertainty-aware training objectives can reduce this type of performance trade-off.</p><p>Together, these findings suggest that accuracy-oriented benchmarks alone may be insufficient for clinical trust alignment. It is not enough for a model to recommend a single correct regimen; it must also demonstrate that the recommendation results from a valid clinical pathway, explicitly ruling out contraindications, preserving relevant patient-specific constraints, and adhering to safety heuristics. Clinical alignment therefore requires evaluation frameworks that jointly assess final-field correctness and the reasoning pathway that produced it.</p></sec><sec id="s4-3"><title>Limitations</title><p>This study has several limitations. First, this is a single-center retrospective dataset from a single reproductive medical center in Chengdu, China. Patient demographics, treatment-strategy preferences, drug availability, and institutional protocols at this center may not generalize to other regions or to resource-constrained settings. The current input schema captures biomedical and reproductive history, but does not include psychological state, treatment preferences, economic constraints, or detailed longitudinal treatment history. This likely limits model performance on psychologically and socioeconomically complex cases. The observed ICSI subgroup decline under GRPO should be interpreted cautiously as a descriptive and hypothesis-generating finding rather than as evidence of a general property of the GRPO algorithm. This pattern may partly reflect the single-center case mix and institution-specific practice style, because ground-truth diagnoses and treatment plans represent &#x201C;single-center ground truth&#x201D;; for clinically plausible cases, different physicians or centers may reasonably annotate different treatment plans based on local protocols, resource availability, and individual treatment philosophy. Beyond these case-mix and annotation-style limitations, the input schema itself was structurally imbalanced: critical male-factor diagnostic parameters such as semen analysis (sperm count, motility, and morphology) were retained only in the unstructured medical-history narrative rather than as structured baseline variables, whereas structured baseline inputs were predominantly female-factor. Because semen analysis is a primary diagnostic criterion for selecting ICSI and related procedures such as TESA+ICSI and IVF with donor sperm, this asymmetry left the model dependent on unstructured-text extraction for male-factor reasoning, and is one plausible contributor to the GRPO performance decline observed in the ICSI subgroup. In addition, because GRPO was trained using a static reward configuration, the observed algorithmic-clinical decoupling may be partly dependent on this specific reward design. Formal reward-weight ablation and external multicenter validation are needed to determine whether the same pattern persists under alternative reward configurations and institutional practice settings. Second, ChatGPT-4o served several compounding roles in this study. It was used to translate unstructured Chinese clinical narratives into English, to generate the CoT reasoning used for SFT, and to serve as the auxiliary semantic judge for the initial-diagnosis field. Therefore, part of the alignment benchmark was model-derived rather than fully independent human-authored clinical logic. This design may introduce teacher-model stylistic bias, residual translation error in reproductive-endocrinology terminology, and evaluation bias toward phrasing patterns preferred by the semantic judge. To mitigate these risks, the CoT prompts were curated by reproductive medicine specialists, the generated reasoning examples were reviewed during dataset construction, and the translation prompts instructed the model to retain medically important information, use professional or standard medical terminology, and translate accurately and concisely; the medical-history prompt additionally specified that the translation should be produced without speculation. In addition, we conducted a bounded bilingual translation audit in which a reproductive medicine specialist reviewed a random subset of 50 translated narratives against the original Chinese records; the audit identified 1 terminology error, 0 numerical errors, 0 omissions or additions, and 0 clinically consequential errors. Although this audit contextualizes the magnitude of translation risk, it does not eliminate the possibility that model-derived reasoning or model-based semantic judging influenced the observed differences between alignment strategies. Future multi-center validation should use independently human-authored reasoning labels, bilingual expert-verified source narratives, and semantic-judge validation against blinded expert adjudication. Third, the blinded independent expert review relied on two reproductive medicine specialists. The 2 reviewers agreed within 1 Likert point on 93.7% of pooled ratings, with PABAK of 0.26 and Gwet AC1 of 0.32 pooled (fair to moderate on the Landis and Koch scale). Raw Cohen &#x03BA; was near zero because rating distributions were strongly concentrated on the upper end of the Likert scale, consistent with the well-described kappa paradox. We mitigate this 3 ways. First, we report Holm-adjusted <italic>P</italic> values (Wilcoxon signed-rank, <italic>P</italic>=.07 for reasoning capability and <italic>P</italic>=.06 for treatment feasibility) and paired effect sizes (Cohen <italic>d</italic>=+0.22 and+0.25, small). Second, the directional pattern is cross-validated by the discrete 3-way best-response preference (SFT 51.2% vs GRPO 26.2% vs charted plan 22.6%), which does not depend on absolute Likert agreement. Third, the field-level subgroup pattern is independently supported by automatic metrics. Expanding the reviewers and including geographically diverse subspecialty experts with a more discriminating evaluation instrument remains an explicit priority for follow-on multicenter validation. Fourth, Likert-scale ratings are ordinal; we additionally report medians and IQRs and note that parametric effect sizes (Cohen <italic>d</italic>) are presented under the assumption of continuous Likert data, which we treat as a study limitation rather than as a claim of interval-level measurement. Fifth, the absolute hallucination rates observed in this study (18.5% for SFT and 15% for GRPO) pose substantial safety risks in a high-stakes domain such as reproductive endocrinology, where a single clinically incorrect or unsupported statement can translate into an inappropriate stimulation protocol, an inappropriate dosage, or an inappropriate procedure selection. Although the postalignment GRPO model showed a relatively lower hallucination rate, the absolute level remains far above any threshold compatible with autonomous or even assistant-mode clinical deployment, and the relative reduction should not be interpreted as evidence of clinical safety. We position all 4 models studied here as research artifacts for evaluating alignment strategy and not as clinical tools. Any future clinical translation of this work should be preceded by safety-critical error grading, calibration analysis, controlled clinician-in-the-loop deployment trials, and explicit fallback or refusal mechanisms when the model&#x2019;s confidence is insufficient. Sixth, the comparison between the model-generated responses and the original physician-charted plans may be affected by format asymmetry. In the reference response, the diagnosis and treatment plan were the genuine physician-charted ground truth, but the accompanying reasoning was not independently human-authored: it was produced by the same ChatGPT-4o&#x2013;based, specialist-curated pipeline used to create the SFT training targets and was physician-verified. All 3 candidate responses therefore shared a similar explicit-reasoning style, so residual presentation-style differences, together with the greater completeness of the model outputs relative to the concise charted plans, may have inflated reviewer preference for the model responses. Any preference for model outputs over charted plans should thus be interpreted as reflecting presentation quality and reasoning completeness rather than clinical superiority over physician decision-making; notably, the SFT baseline was preferred even more often than the reference response itself (51.2% vs 22.6%). This also means the alignment benchmark is in part model-derived. The SFT-versus-GRPO comparison is less affected by this limitation because both models generated outputs in the same format. Finally, the ICL evaluation relied on a single gated prompt configuration that asked the model to consult the guideline only when uncertain about its answer. Because contemporary LLMs are known to calibrate epistemic uncertainty poorly, the underperformance of ICL relative to SFT in our study should be attributed to this specific prompt implementation rather than to the ICL paradigm in general. Systematic comparison against unconditional guideline conditioning, retrieval-augmented prompting, and chain-of-verification approaches remains a priority for follow-on work.</p></sec><sec id="s4-4"><title>Future Directions</title><p>Translating these findings into clinical practice will require a phased validation pathway. The immediate next step is external multi-center prospective validation across centers with different case-mix and protocol preferences. A second step is integration with electronic health record systems so that model reasoning chains can be inspected at the point of care, rather than after the fact. A third step is testing dynamic decision-making in longitudinal scenarios such as follicle monitoring, where treatment decisions are revised as new information arrives. Additionally, incorporate multimodal patient-profile inputs, including psychological state, treatment-preference elicitation, socioeconomic context, and longitudinal treatment history, to build more comprehensive ART decision-support systems. Beyond the decision-support role studied here, future iterations should also incorporate referral markers (eg, automatically recommending genetic counseling for patients with a history of recurrent miscarriage) so that the system reflects the multidisciplinary structure of reproductive medicine care. Methodologically, future work should pair process-level and outcome-level rewards within a single training objective, conduct ablation studies on reward weighting, and extend evaluation dimensions of &#x201C;treatment feasibility&#x201D; into operational subdimensions such as drug availability, monitoring burden, and the impact on patients&#x2019; work and personal life. Future reward design should also incorporate structured ART guidelines and clinical constraints, so that inconsistent reasoning steps, unsupported causal jumps, contraindication violations, and deviations from established treatment pathways can be explicitly penalized rather than ignored by final-answer scoring. Future evaluation frameworks should also move beyond overall accuracy to include fine-grained clinical fairness and subgroup robustness. Clinically important margin cases, such as stimulation risk profiles, embryo-yield expectations, dosing safety margins, and adherence to OHSS-prevention heuristics, may provide a more operationally meaningful assessment of whether model recommendations remain safe and useful near decision boundaries. Finally, future studies should examine whether the Alignment Paradox persists across model scales. Larger medical LLM backbones may encode richer clinical context and may differ in their sensitivity to reward-induced distributional shifts, but this remains an empirical question rather than an assumption.</p></sec><sec id="s4-5"><title>Conclusions</title><p>In this single-center retrospective study, automatic metrics were compared across four alignment paradigms, whereas clinician-perceived reasoning quality was evaluated in the SFT versus GRPO contrast. GRPO achieved the strongest automatic field-level performance, while SFT showed directionally higher expert ratings for reasoning capability and treatment feasibility, and the absolute hallucination rate of both models remained too high for clinical deployment. Outcome-based metrics alone are therefore insufficient proxies for clinical utility in ART AI systems, and external validation paired with process-level evaluation is required before any deployment claim can be supported. Because the clinical review was based on two reproductive medicine specialists with marginal inter-rater agreement, conclusions about clinical alignment should be interpreted as exploratory and require confirmation in larger multi-center expert panels.</p></sec></sec></body><back><ack><p>The authors thank the clinical staff of the Reproductive Medical Center, West China Second University Hospital, Sichuan University, for their support in data curation and clinical consultation. The authors also thank the 2 reproductive medicine specialists who completed the independent blinded review described in the Methods section.</p><p>During the preparation of this manuscript, the authors used Grammarly to correct grammatical errors and typos. ChatGPT-4o was also used during the study itself (and is described in the Methods) for (1) generating chain-of-thought reasoning during dataset curation and (2) translating Chinese clinical narratives into English; no generative AI tool was used to draft scientific content or interpretation in the manuscript text. After using these tools, the authors reviewed and edited all content as needed and take full responsibility for the content of the publication.</p></ack><notes><sec><title>Funding</title><p>Part of this work was supported in part by the Science and Technology Department of Sichuan Province Project (2024YFFK0365); Part of this study was supported by the Natural Science Foundation of Sichuan, China (2025NSFSC1985); Part of this study was supported by the 1&#x00B7;3&#x00B7;5 project for disciplines of excellence, West China Hospital, Sichuan University (ZYYC21004).</p></sec><sec><title>Data Availability</title><p>The clinical dataset analyzed in this study contains de-identified electronic health records from West China Second University Hospital, Sichuan University, and cannot be made publicly available because of institutional patient-privacy and data-protection policies. De-identified excerpts of model inputs and outputs sufficient to reproduce the reported analyses are available from the corresponding author upon reasonable request and subject to approval by the Ethics Committee of West China Second University Hospital. All training and evaluation code, prompt templates, and model checkpoints required to reproduce the main results are available by request.</p></sec></notes><fn-group><fn fn-type="con"><p>Clinical validation: Y Long</p><p>Conceptualization: Dou Liu, Y Long, R Yin</p><p>Data curation: Dou Liu, SZ</p><p>Formal analysis: Dou Liu, SZ</p><p>Funding acquisition: KL, R Yin</p><p>Investigation: Dou Liu, KX, R Yang, Y Lin, HL</p><p>Methodology: Dou Liu, Y Long</p><p>Project administration: R Yin</p><p>Resources: KL, R Yin</p><p>Software: Dou Liu, SZ, Di Liu, KX, R Yang</p><p>Supervision: KL, R Yin</p><p>Validation: Di Liu, Y Lin, HL</p><p>Visualization: Dou Liu, Di Liu</p><p>Writing &#x2013; original draft: Dou Liu, Y Long</p><p>Writing &#x2013; review &#x0026; editing: Dou Liu, Y Long, KL, R Yin</p><p>All authors have read and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ART</term><def><p>assisted reproductive technology</p></def></def-item><def-item><term id="abb2">CC</term><def><p>clomiphene citrate</p></def></def-item><def-item><term id="abb3">COS</term><def><p>controlled ovarian stimulation</p></def></def-item><def-item><term id="abb4">CoT</term><def><p>chain-of-thought</p></def></def-item><def-item><term id="abb5">DPO</term><def><p>direct preference optimization</p></def></def-item><def-item><term id="abb6">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb7">GnRH</term><def><p>gonadotropin-releasing hormone</p></def></def-item><def-item><term id="abb8">GRPO</term><def><p>group relative policy optimization</p></def></def-item><def-item><term id="abb9">HIPAA</term><def><p>Health Insurance Portability and Accountability Act</p></def></def-item><def-item><term id="abb10">ICL</term><def><p>in-context learning</p></def></def-item><def-item><term id="abb11">ICSI</term><def><p>intracytoplasmic sperm injection</p></def></def-item><def-item><term id="abb12">IVF</term><def><p>in vitro fertilization</p></def></def-item><def-item><term id="abb13">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb14">LoRA</term><def><p>low-rank adaptation</p></def></def-item><def-item><term id="abb15">MAE</term><def><p>mean absolute error</p></def></def-item><def-item><term id="abb16">OHSS</term><def><p>ovarian hyperstimulation syndrome</p></def></def-item><def-item><term id="abb17">PABAK</term><def><p>prevalence-adjusted bias-adjusted kappa</p></def></def-item><def-item><term id="abb18">PGT</term><def><p>preimplantation genetic testing</p></def></def-item><def-item><term id="abb19">PGT-A</term><def><p>preimplantation genetic testing for aneuploidy</p></def></def-item><def-item><term id="abb20">PGT-M</term><def><p>preimplantation genetic testing for monogenic disorders</p></def></def-item><def-item><term id="abb21">PGT-SR</term><def><p>preimplantation genetic testing for structural rearrangements</p></def></def-item><def-item><term id="abb22">PPO</term><def><p>proximal policy optimization</p></def></def-item><def-item><term id="abb23">PPOS</term><def><p>progestin-primed ovarian stimulation</p></def></def-item><def-item><term id="abb24">RLHF</term><def><p>reinforcement learning from human feedback</p></def></def-item><def-item><term id="abb25">RQ</term><def><p>research question</p></def></def-item><def-item><term id="abb26">SFT</term><def><p>supervised fine-tuning</p></def></def-item><def-item><term id="abb27">SOTA</term><def><p>state-of-the-art</p></def></def-item><def-item><term id="abb28">TESA</term><def><p>testicular sperm aspiration</p></def></def-item><def-item><term id="abb29">WHO</term><def><p>World Health Organization</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="report"><article-title>Infertility prevalence estimates, 1990&#x2013;2021</article-title><year>2023</year><access-date>2026-09-12</access-date><publisher-name>World Health Organization</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/publications/i/item/978920068315">https://www.who.int/publications/i/item/978920068315</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Graham</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Jelin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hoon</surname><given-names>AH</given-names>  <suffix>Jr</suffix></name><name name-style="western"><surname>Wilms Floet</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Levey</surname><given-names>E</given-names> </name><name name-style="western"><surname>Graham</surname><given-names>EM</given-names> </name></person-group><article-title>Assisted reproductive technology: short&#x2010; and long&#x2010;term outcomes</article-title><source>Develop Med Child Neuro</source><year>2023</year><month>01</month><volume>65</volume><issue>1</issue><fpage>38</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1111/dmcn.15332</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kumar</surname><given-names>P</given-names> </name><name name-style="western"><surname>Sait</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>M</given-names> </name></person-group><article-title>Ovarian hyperstimulation syndrome</article-title><source>J Hum Reprod Sci</source><year>2011</year><month>05</month><volume>4</volume><issue>2</issue><fpage>70</fpage><lpage>75</lpage><pub-id pub-id-type="doi">10.4103/0974-1208.86080</pub-id><pub-id pub-id-type="medline">22065820</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zheng</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kwok</surname><given-names>E</given-names> </name><name name-style="western"><surname>Taljaard</surname><given-names>M</given-names> </name><name name-style="western"><surname>Nemnom</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Stiell</surname><given-names>I</given-names> </name></person-group><article-title>Decision fatigue in the emergency department: how does emergency physician decision making change over an eight-hour shift?</article-title><source>Am J Emerg Med</source><year>2020</year><month>12</month><volume>38</volume><issue>12</issue><fpage>2506</fpage><lpage>2510</lpage><pub-id pub-id-type="doi">10.1016/j.ajem.2019.12.020</pub-id><pub-id pub-id-type="medline">31937441</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="other"><person-group person-group-type="author"><collab>DeepSeek-AI</collab><name name-style="western"><surname>Guo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>DeepSeek-R1: incentivizing reasoning capability in llms via reinforcement learning</article-title><source>arXiv</source><comment>Preprint posted online on  Jan, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.12948</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Rafailov</surname><given-names>R</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mitchell</surname><given-names>E</given-names> </name><name name-style="western"><surname>Manning</surname><given-names>CD</given-names> </name><name name-style="western"><surname>Ermon</surname><given-names>S</given-names> </name><name name-style="western"><surname>Finn</surname><given-names>C</given-names> </name></person-group><article-title>Direct preference optimization: your language model is secretly a reward model</article-title><year>2023</year><month>12</month><day>15</day><conf-name>Advances in Neural Information Processing Systems 36</conf-name><conf-loc>New Orleans, Louisiana, USA</conf-loc><fpage>53728</fpage><lpage>53741</lpage><pub-id pub-id-type="doi">10.52202/075280-2338</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lai</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhong</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Med-R1: reinforcement learning for generalizable medical reasoning in vision-language models</article-title><source>IEEE Trans Med Imaging</source><year>2025</year><month>03</month><volume>45</volume><issue>6</issue><fpage>2727</fpage><lpage>2737</lpage><pub-id pub-id-type="doi">10.1109/TMI.2026.3661001</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Shao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>DeepSeekMath: pushing the limits of mathematical reasoning in open language models</article-title><source>arXiv</source><comment>Preprint posted online on  Feb, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2402.03300</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Dao</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vu</surname><given-names>DB</given-names> </name></person-group><article-title>AlphaMaze: enhancing large language models&#x2019; spatial intelligence via GRPO</article-title><source>arXiv</source><comment>Preprint posted online on  Feb, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2502.14669</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>K</given-names> </name><etal/></person-group><article-title>HuatuoGPT-o1, towards medical complex reasoning with LLMs</article-title><source>arXiv</source><comment>Preprint posted online on  Dec, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.18925</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gottweis</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="medline">39779926</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Team</surname><given-names>L</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>HP</given-names> </name><etal/></person-group><article-title>Lingshu: a generalist foundation model for unified multimodal medical understanding and reasoning</article-title><source>arXiv</source><comment>Preprint posted online on  Jun, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2506.07044</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gaber</surname><given-names>F</given-names> </name><name name-style="western"><surname>Shaik</surname><given-names>M</given-names> </name><name name-style="western"><surname>Allega</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Evaluating large language model workflows in clinical decision support for triage and referral and diagnosis</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>9</day><volume>8</volume><issue>1</issue><fpage>263</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01684-1</pub-id><pub-id pub-id-type="medline">40346344</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Tian</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Automated procedural analysis via video-language models for AI-assisted nursing skills assessment</article-title><source>IISE Transactions on Healthcare Systems Engineering</source><year>2026</year><month>06</month><day>5</day><volume>16</volume><issue>2</issue><fpage>210</fpage><lpage>223</lpage><pub-id pub-id-type="doi">10.1080/24725579.2026.2651352</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Han</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Jin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Kong</surname><given-names>YK</given-names> </name><name name-style="western"><surname>Park</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yun</surname><given-names>MH</given-names> </name></person-group><article-title>Evaluating the application of chatgpt in outpatient triage guidance: a comparative study</article-title><source>Proc 22nd Congr Int Ergon Assoc</source><year>2025</year><fpage>233</fpage><lpage>238</lpage><pub-id pub-id-type="doi">10.1007/978-981-95-0211-0_36</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Long</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zuoqiu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>R</given-names> </name></person-group><article-title>Evaluating the feasibility and accuracy of large language models for medical history-taking in obstetrics and gynecology</article-title><source>arXiv</source><comment>Preprint posted online on  Mar, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.00061</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>N</given-names> </name></person-group><article-title>Large language models in health care: development, applications, and challenges</article-title><source>Health Care Sci</source><year>2023</year><month>08</month><volume>2</volume><issue>4</issue><fpage>255</fpage><lpage>263</lpage><pub-id pub-id-type="doi">10.1002/hcs2.61</pub-id><pub-id pub-id-type="medline">38939520</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hager</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jungmann</surname><given-names>F</given-names> </name><name name-style="western"><surname>Holland</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Evaluation and mitigation of the limitations of large language models in clinical decision-making</article-title><source>Nat Med</source><year>2024</year><month>09</month><volume>30</volume><issue>9</issue><fpage>2613</fpage><lpage>2622</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03097-1</pub-id><pub-id pub-id-type="medline">38965432</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>MedChain: bridging the gap between LLM agents and clinical practice with interactive sequence</article-title><conf-name>Advances in Neural Information Processing Systems 38</conf-name><conf-date>Dec 2-7, 2025</conf-date><pub-id pub-id-type="doi">10.52202/085713-2498</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ullah</surname><given-names>E</given-names> </name><name name-style="western"><surname>Parwani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Baig</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>R</given-names> </name></person-group><article-title>Challenges and barriers of using large language models (LLM) such as ChatGPT for diagnostic medicine with a focus on digital pathology - a recent scoping review</article-title><source>Diagn Pathol</source><year>2024</year><month>02</month><day>27</day><volume>19</volume><issue>1</issue><fpage>43</fpage><pub-id pub-id-type="doi">10.1186/s13000-024-01464-7</pub-id><pub-id pub-id-type="medline">38414074</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dwivedi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Dave</surname><given-names>D</given-names> </name><name name-style="western"><surname>Naik</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Explainable AI (XAI): core ideas, techniques, and solutions</article-title><source>ACM Comput Surv</source><year>2023</year><month>09</month><day>30</day><volume>55</volume><issue>9</issue><fpage>1</fpage><lpage>33</lpage><pub-id pub-id-type="doi">10.1145/3561048</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Evans</surname><given-names>T</given-names> </name><name name-style="western"><surname>Retzlaff</surname><given-names>CO</given-names> </name><name name-style="western"><surname>Gei&#x00DF;ler</surname><given-names>C</given-names> </name><etal/></person-group><article-title>The explainability paradox: challenges for xAI in digital pathology</article-title><source>Future Generation Computer Systems</source><year>2022</year><month>08</month><volume>133</volume><fpage>281</fpage><lpage>296</lpage><pub-id pub-id-type="doi">10.1016/j.future.2022.03.009</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mart&#x00ED;nez-Ag&#x00FC;ero</surname><given-names>S</given-names> </name><name name-style="western"><surname>Soguero-Ruiz</surname><given-names>C</given-names> </name><name name-style="western"><surname>Alonso-Moral</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Mora-Jim&#x00E9;nez</surname><given-names>I</given-names> </name><name name-style="western"><surname>&#x00C1;lvarez-Rodr&#x00ED;guez</surname><given-names>J</given-names> </name><name name-style="western"><surname>Marques</surname><given-names>AG</given-names> </name></person-group><article-title>Interpretable clinical time-series modeling with intelligent feature selection for early prediction of antimicrobial multidrug resistance</article-title><source>Future Generation Computer Systems</source><year>2022</year><month>08</month><volume>133</volume><fpage>68</fpage><lpage>83</lpage><pub-id pub-id-type="doi">10.1016/j.future.2022.02.021</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bosma</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>VY</given-names> </name><etal/></person-group><article-title>Finetuned language models are zero-shot learners</article-title><source>arXiv</source><comment>Preprint posted online on  Sep, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2109.01652</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yue</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>W</given-names> </name></person-group><article-title>Critique fine-tuning: learning to critique is more effective than learning to imitate</article-title><source>arXiv</source><comment>Preprint posted online on  Jan, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.17703</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dai</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ekbote</surname><given-names>C</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>P</given-names> </name></person-group><article-title>QoQ-Med: building multimodal clinical foundation models with domain-aware GRPO training</article-title><conf-name>Advances in Neural Information Processing Systems 38</conf-name><conf-date>Dec 2-7, 2025</conf-date><pub-id pub-id-type="doi">10.52202/085713-1256</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Long</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zuoqiu</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Reliability of large language model generated clinical reasoning in assisted reproductive technology: blinded comparative evaluation study</article-title><source>J Med Internet Res</source><year>2026</year><month>01</month><day>8</day><volume>28</volume><issue>1</issue><fpage>e85206</fpage><pub-id pub-id-type="doi">10.2196/85206</pub-id><pub-id pub-id-type="medline">41505193</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="web"><article-title>OpenBioLLM-8B</article-title><source>Hugging Face</source><access-date>2025-07-29</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/aaditya/Llama3-OpenBioLLM-8B">https://huggingface.co/aaditya/Llama3-OpenBioLLM-8B</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>AlShibli</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bazi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Rahhal</surname><given-names>MMA</given-names> </name><name name-style="western"><surname>Zuair</surname><given-names>M</given-names> </name></person-group><article-title>Vision-BioLLM: large vision language model for visual dialogue in biomedical imagery</article-title><source>Biomed Signal Process Control</source><year>2025</year><month>05</month><volume>103</volume><fpage>107437</fpage><pub-id pub-id-type="doi">10.1016/j.bspc.2024.107437</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yuan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nie</surname><given-names>M</given-names> </name></person-group><article-title>Fine-tuning a personalized OpenBioLLM using offline reinforcement learning</article-title><source>Applied Sciences</source><year>2025</year><volume>15</volume><issue>5</issue><fpage>2486</fpage><pub-id pub-id-type="doi">10.3390/app15052486</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Schulman</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wolski</surname><given-names>F</given-names> </name><name name-style="western"><surname>Dhariwal</surname><given-names>P</given-names> </name><name name-style="western"><surname>Radford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Klimov</surname><given-names>O</given-names> </name></person-group><article-title>Proximal policy optimization algorithms</article-title><source>arXiv</source><comment>Preprint posted online on  Jul, 2017</comment><pub-id pub-id-type="doi">10.48550/arXiv.1707.06347</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Schulman</surname><given-names>J</given-names> </name><name name-style="western"><surname>Moritz</surname><given-names>P</given-names> </name><name name-style="western"><surname>Levine</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jordan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Abbeel</surname><given-names>P</given-names> </name></person-group><article-title>High-dimensional continuous control using generalized advantage estimation</article-title><source>arXiv</source><comment>Preprint posted online on  Jun, 2018</comment><pub-id pub-id-type="doi">10.48550/arXiv.1506.02438</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>In-context learning guideline blocks, reproducibility details, direct preference optimization epoch sensitivity analysis, and full doctor-in-the-loop evaluation statistics.</p><media xlink:href="jmir_v28i1e97221_app1.docx" xlink:title="DOCX File, 35 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Doctor-in-the-loop evaluation. (A) Expert-rated performance across three dimensions (5-point Likert scale and mean of 2 raters). Supervised fine-tuning (SFT) shows consistently higher reasoning and feasibility scores (Holm-adjusted <italic>P</italic>=.07 and .06, respectively); group relative policy optimization (GRPO) has a descriptively lower hallucination percentage. (B) Blinded best-response selection across SFT, GRPO, and physician-charted plans (GT). SFT won more often than both GRPO (<italic>P</italic>&#x003C;.001) and GT (<italic>P</italic>&#x003C;.001); the difference between GRPO and GT was not statistically significant (<italic>P</italic>=.48).</p><media xlink:href="jmir_v28i1e97221_app2.png" xlink:title="PNG File, 245 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Subgroup performance across 4 alignment strategies and 3 major assisted reproductive technology categories: in vitro fertilization (IVF), intracytoplasmic sperm injection (ICSI), preimplantation genetic testing (PGT), in the held-out test set (single-center retrospective dataset). <italic>F</italic><sub>1</sub>-scores are displayed on a 0-1 scale. Group relative policy optimization improved <italic>F</italic><sub>1</sub> in IVF and PGT but reduced <italic>F</italic><sub>1</sub> in ICSI, where male-factor diagnostic parameters were captured primarily in unstructured fields.</p><media xlink:href="jmir_v28i1e97221_app3.png" xlink:title="PNG File, 49 KB"/></supplementary-material></app-group></back></article>