<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e92919</article-id><article-id pub-id-type="doi">10.2196/92919</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Prompt Configurations for Multimodal Large Language Models in Diagnosing and Staging Osteonecrosis of the Femoral Head: Multimodel Retrospective Observational Diagnostic Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Zhu</surname><given-names>Jiesheng</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Huang</surname><given-names>Xingxing</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Shi</surname><given-names>Jincheng</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Shaoming</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Daosen</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gao</surname><given-names>Zhihan</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lu</surname><given-names>Yi</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yang</surname><given-names>Tao</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Xue</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lin</surname><given-names>Yimu</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Xu</surname><given-names>Peiyu</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Li</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Fan</surname><given-names>Pei</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Orthopedics, Second Affiliated Hospital &#x0026; Yuying Children's Hospital of Wenzhou Medical University</institution><addr-line>No. 109, Xueyuan West Road</addr-line><addr-line>Wenzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Orthopedics, Yunhe People's Hospital</institution><addr-line>Lishui City</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><aff id="aff3"><institution>The Second School of Medicine, Wenzhou Medical University</institution><addr-line>Wenzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><aff id="aff4"><institution>Department of Radiology, Second Affiliated Hospital &#x0026; Yuying Children's Hospital of Wenzhou Medical University</institution><addr-line>Wenzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><aff id="aff5"><institution>Key Laboratory of Pediatric Anesthesiology, Ministry of Education, Wenzhou Medical University</institution><addr-line>Wenzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Zhang</surname><given-names>Jun</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Liang</surname><given-names>Xiaolong</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Yu</surname><given-names>Yunguo</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Pei Fan, MD, Department of Orthopedics, Second Affiliated Hospital &#x0026; Yuying Children's Hospital of Wenzhou Medical University, No. 109, Xueyuan West Road, Wenzhou, Zhejiang, 325000, China, 86 577 88002808, 86 577 88002823; <email>fanpei@wmu.edu.cn</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>10</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e92919</elocation-id><history><date date-type="received"><day>05</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>24</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>25</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Jiesheng Zhu, Xingxing Huang, Jincheng Shi, Shaoming Chen, Daosen Chen, Zhihan Gao, Yi Lu, Tao Yang, Xue Wang, Yimu Lin, Peiyu Xu, Li Li, Pei Fan. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 10.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e92919"/><abstract><sec><title>Background</title><p>Multimodal large language models (MLLMs) have emerging potential for interpreting medical images and text, but their performance in orthopedic imaging tasks and the influence of prompt configuration remain insufficiently studied.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the performance of commercial and open-source MLLMs for diagnosing and staging osteonecrosis of the femoral head (ONFH) and to assess how different prompt configurations affect model performance.</p></sec><sec sec-type="methods"><title>Methods</title><p>This single-center retrospective diagnostic accuracy study included 159 radiograph patients contributing 318 hip-level observations and 170 magnetic resonance imaging (MRI) patients contributing 340 hip-level observations; 55 patients with both modalities formed the multi-image (MI) subgroup between July 2023 and December 2024. Four MLLMs were evaluated: GPT-4o, Claude 3.7 Sonnet, Qwen2.5-VL 72B, and Gemma 3 27B. Three prompt configurations were tested: single image (SI), image plus radiology description (ID), and MI. Model performance was assessed for ONFH detection; early- versus late-stage differentiation; detailed grading using the Ficat, Association Research Circulation Osseous (ARCO), and Steinberg systems; and grading reliability using intraclass correlation coefficients (ICCs).</p></sec><sec sec-type="results"><title>Results</title><p>Model performance varied by prompt configuration and imaging input. For ONFH detection, the SI configuration yielded a mean detection area under the receiver operating characteristic curve (AUC) of 0.55 (SD 0.03), whereas the ID configuration achieved a mean detection AUC of 0.91 (SD 0.01). In radiograph-based ONFH detection, ID input achieved a mean accuracy of 0.88 (SD 0.01); in MRI-based ONFH detection, ID input achieved a mean accuracy of 0.85 (SD 0.01). For early- versus late-stage differentiation, the mean accuracy was 0.65 (SD 0.11) with SI input, 0.78 (SD 0.04) with ID input, and approximately 0.59 (SD 0.10) with MI input. For detailed grading, ID input improved mean accuracy across the Ficat, ARCO, and Steinberg systems compared with SI input. In ARCO grading reliability analysis, the mean MLLM ICC was 0.51 (SD 0.18) for SI, 0.97 (SD 0.02) for ID, and 0.49 (SD 0.15) for MI; surgeon interrater and intrarater ICCs were 0.70 and 0.81, respectively. In the commercial versus open-source model comparison, no significant overall difference was observed between model groups (<italic>P</italic>=.83).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Prompt configuration strongly influenced MLLM performance in ONFH diagnosis and staging. Pairing images with deidentified radiology descriptions improved diagnostic and grading performance, whereas MI input did not provide consistent additional benefit in this retrospective single-center evaluation. These findings support the potential role of MLLMs as assistive tools in human-AI orthopedic imaging workflows, but external validation, careful input standardization, and prospective clinical evaluation are needed before clinical deployment.</p></sec></abstract><kwd-group><kwd>osteonecrosis of the femoral head</kwd><kwd>artificial intelligence</kwd><kwd>multimodal large language models</kwd><kwd>radiology</kwd><kwd>diagnostic imaging</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Osteonecrosis of the femoral head (ONFH) is a common progressive musculoskeletal condition primarily affecting young and middle-aged adults [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>], with over 20,000 new cases annually in the United States [<xref ref-type="bibr" rid="ref3">3</xref>]. Early ONFH is often asymptomatic and bilateral; without timely intervention, over half of affected hips may collapse and require total hip arthroplasty (THA) within a few years [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. Given that THA in young patients is associated with higher revision rates and substantial long-term health care costs, accurate early diagnosis and precise staging are critical for guiding joint-preserving treatments, delaying disease progression, and improving long-term functional outcomes.</p><p>Radiographic assessment remains the cornerstone of ONFH diagnosis, but standard X-rays lack sensitivity in early disease [<xref ref-type="bibr" rid="ref6">6</xref>]. Magnetic resonance imaging (MRI) improves early detection but provides limited assessment of late-stage collapse. Widely used grading systems, such as Ficat, Association Research Circulation Osseous (ARCO), and Steinberg, help standardize evaluation [<xref ref-type="bibr" rid="ref7">7</xref>], but they differ in criteria and are subject to considerable interobserver variability [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. These limitations highlight the need for diagnostic tools that can integrate multimodal information and enhance consistency across readers.</p><p>AI is a promising tool, with multimodal large language models (MLLMs) capable of processing both imaging and textual data, recently emerged as powerful tools in medical imaging [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. Unlike traditional deep learning (DL) models trained for single diagnosis tasks [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref16">16</xref>], MLLMs can integrate imaging features with clinical context, perform higher-order reasoning, and generate structured reports [<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref19">19</xref>]. These capabilities position MLLMs as potential assistants for diagnostic interpretation in complex clinical workflows.</p><p>Despite their promise, MLLM applications in clinical practice still face challenges [<xref ref-type="bibr" rid="ref20">20</xref>]. In particular, limited translational research from computational models to clinical applications constrains the practical implementation of these advanced technologies. Prior studies have largely focused on testing commercial models and single-image tasks [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Commercial systems face challenges in cost, accessibility, and adaptability, whereas open-source models may offer more feasible alternatives for clinical adoption. To date, no systematic study has compared commercial and open-source MLLMs in complex orthopedic diagnostic tasks. In addition, the exploration of prompt engineering design for diagnosis has been limited&#x2014;particularly in combining imaging with clinical text or integrating multimodal inputs. It remains unclear whether optimized prompting strategies can enhance model performance and increase clinical use. These gaps underscore the need to test the full clinical workflow.</p><p>To address these gaps, we systematically explored the potential of MLLMs in clinical applications. We evaluated 4 advanced models&#x2014;2 commercial (ChatGPT-4o, Claude 3.7) and 2 open-source (Qwen2.5-VL, Gemma-3)&#x2014;on core ONFH-related tasks: diagnosis, early- or late-stage differentiation, and staging. Furthermore, we examined the effect of prompt configuration by comparing single image (SI), image plus radiology description (ID), and multi-image (MI) inputs across different grading systems with quantitative metrics.</p><p>Therefore, the objective of this study is to evaluate, from a clinical observation perspective, whether MLLMs can serve as an assistive tool for diagnosing and staging ONFH within a human-machine collaborative workflow. Specifically, we aim to assess (1) whether MLLMs can diagnose and stage ONFH across commonly used classification systems, (2) whether different models (particularly open-source models) exhibit performance variations in clinical diagnosis and staging, and (3) whether prompting designs for different clinical contexts can enhance model performance and reliability.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Ethical Considerations</title><p>This retrospective study was reviewed and approved by the Ethics Committee of the Second Affiliated Hospital and Yuying Children&#x2019;s Hospital of Wenzhou Medical University (approval number 2025-K-282&#x2010;01; approval date: February 8, 2025). The approval was obtained before any study-specific research procedures were initiated, including data extraction for this study, data deidentification, MLLM evaluation, and statistical analysis. Eligible cases were retrospectively identified from historical imaging data generated during routine clinical care between July 2023 and December 2024. Because this study involved only retrospective analysis of deidentified clinical imaging data and did not involve active patient intervention, the requirement for written informed consent was waived by the ethics committee. All personal identifiers were removed before analysis, and all data were handled in accordance with institutional privacy and confidentiality requirements. The requirement for written informed consent was waived by the Ethics Committee because this retrospective study used deidentified clinical imaging data and involved no active patient intervention.</p></sec><sec id="s2-2"><title>Data Collection and Preparation</title><p>This was a single-center retrospective diagnostic accuracy study conducted at the Second Affiliated Hospital and Yuying Children&#x2019;s Hospital of Wenzhou Medical University, a tertiary referral hospital in Wenzhou, China. Potentially eligible cases were systematically identified from the in-house INFINITY picture archiving and communication system (PACS). The search included patients who underwent standardized anteroposterior pelvic radiographs or multisequence pelvic MRI for suspected or confirmed ONFH between July 2023 and December 2024. MRI examinations included at least T1-weighted imaging (T1WI) and T2-weighted fat-suppressed (T2FS) sequences.</p><p>Cases were screened in a stepwise manner. First, imaging records during the study period were retrieved from the PACS according to modality and clinical indication. Second, eligibility was assessed based on the predefined inclusion and exclusion criteria listed in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Third, 2 senior radiologists reviewed image quality and diagnostic relevance (one with 15 years of experience and the other with 17 years of experience). Cases with incomplete imaging, severe artifacts, insufficient field of view, prior hip arthroplasty, or other findings that precluded reliable ONFH grading were excluded. For eligible MRI cases, 2 radiologists jointly selected representative slices for each patient, prioritizing images showing (1) maximal lesion diameter, (2) key diagnostic features, and (3) optimal lesion-to-marrow contrast. All eligible cases meeting the criteria during the study period were included consecutively; no random sampling was performed.</p><p>This single-center retrospective diagnostic accuracy study included 159 radiographic patients contributing 318 hip-level observations and 170 MRI patients contributing 340 hip-level observations; 55 patients with both modalities formed the MI subgroup. The distributions by disease status and laterality are detailed as follows: radiographs (18 non-ONFH, 141 ONFH [98 unilateral, 43 bilateral]) and MRI (24 non-ONFH, 146 ONFH [84 unilateral, 62 bilateral]).</p></sec><sec id="s2-3"><title>Imaging Grading Systems</title><p>To test the model&#x2019;s generalization ability, 3 established imaging grading systems were selected for ONFH classification: the Ficat system [<xref ref-type="bibr" rid="ref22">22</xref>], the ARCO system [<xref ref-type="bibr" rid="ref23">23</xref>], and the Steinberg system [<xref ref-type="bibr" rid="ref24">24</xref>]. The specific criteria for each system are detailed in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Briefly, the Ficat system primarily emphasizes radiographic changes and femoral head collapse, the ARCO system incorporates radiographic and MRI findings with subcategorization by femoral head depression, and the Steinberg system provides a more quantitative severity framework based on lesion extent and structural progression. ARCO staging was operationalized using the radiographic and MRI criteria available in this study. Bone scintigraphy was not performed or used for reference-standard assignment. In the radiograph-only cohort, no hip was assigned to ARCO stage 1; radiographically normal hips were classified as ARCO stage 0. This operational category indicated the absence of radiographic abnormalities and did not correspond to asymptomatic ARCO stage 0 in the published 2019 ARCO criteria. In the MRI cohort, ARCO stage 1 was assigned when radiographs were normal, but MRI showed findings compatible with ONFH, such as a band-like low-signal-intensity lesion around the necrotic area. Because bone scintigraphy was unavailable for most patients, the ARCO results should be interpreted as radiograph or MRI-based operationalized ARCO staging rather than strict application of the complete 2019 ARCO staging system.</p></sec><sec id="s2-4"><title>Surgeon Grading and Reference Standard Establishment</title><p>Two senior orthopedic surgeons (A: 18 years of experience; B: 15 years of experience) and 1 junior orthopedic surgeon (C: 5 years of experience) independently graded all images using the Ficat, ARCO, and Steinberg systems. Their results were used for comparison with MLLM performance.</p><p>The reference standard was established independently at the hip level by a panel of another 4 experts, comprising 2 musculoskeletal radiologists and 2 additional senior orthopedic surgeons (average of 20 years of experience). Each panel member independently evaluated every hip using the prespecified diagnostic and staging criteria and was blinded to the MLLM outputs and the assessments of the comparison surgeons. Individual assessments were recorded before the consensus discussion. The mean interrater intraclass correlation coefficient (ICC) across the 3 grading systems was 0.84 (SD 0.05). Disagreements were resolved through panel discussion, and the final consensus diagnosis and stage for each hip served as the reference standard.</p></sec><sec id="s2-5"><title>Selection of MLLMs</title><p>Model selection was based on availability and reported multimodal capabilities when model evaluation began in April 2025. GPT-4o (released May 13, 2024) and Claude 3.7 Sonnet (released February 24, 2025) were selected as representative leading commercial models with strong reported image understanding and multimodal reasoning capabilities. Qwen2.5-VL-72B-Instruct (released January 26, 2025) and Gemma 3 27B instruction-tuned (released March 12, 2025) were selected as recently released, high-capacity open-source vision-language models. The largest available variants of these open-source model families were selected to maximize their potential visual reasoning capabilities.</p><p>Model evaluations began in April 2025 and were conducted using GPT-4o (model ID: gpt-4o-2024-11-20), Claude 3.7 Sonnet (model ID: claude-3&#x2010;7-sonnet-20250219), Qwen2.5-VL-72B-Instruct (model ID: qwen2.5-vl-72b-instruct), and Gemma 3 27B instruction-tuned (checkpoint/model ID: google/gemma-3-27b-it). All 4 models were evaluated during the same study period using standardized prompts and image inputs.</p></sec><sec id="s2-6"><title>Prompt Design</title><p>To address diagnostic refusal issues and complete clinical tasks in a standardized format, we developed a systematic prompt strategy (<xref ref-type="fig" rid="figure1">Figure 1B</xref>), including a system prompt and a user prompt. A detailed prompt framework and representative radiology description examples are shown in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study workflow and multimodal prompt configurations. (A) Multimodal large language model (MLLM) testing workflow. (B) Grading prompt design framework and image description example. (C) Commercial and open-source MLLMs evaluated. (D) Multimodal prompt configurations: single image (SI), image with description (ID), and multiple images (MI). MI included setup 1 (X-ray + magnetic resonance imaging [MRI] slices) and setup 2 (MRI slices only). MIS1: multi-image setup 1; MIS2: multi-image setup 2; T1WI: T1-weighted imaging; T2FS: T2-weighted fat-suppressed sequences.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92919_fig01.png"/></fig></sec><sec id="s2-7"><title>Multimodal Prompt Configuration Design for MLLMs</title><p>To investigate whether different multimodal prompt configurations impact performance and enhance clinical utility, we evaluated 2 strategies in prompt engineering. Three multimodal prompt configurations (<xref ref-type="fig" rid="figure1">Figure 1D</xref>) were tested to assess MLLM performance across input types:</p><list list-type="order"><list-item><p>SI: This configuration establishes a baseline for the model&#x2019;s zero-shot visual inference capabilities and tests the model&#x2019;s ability to diagnose and grade using a single medical image per patient. The tests included independent evaluations using one of the following inputs: (1) 1 anteroposterior pelvic radiograph, (2) 1 representative T1WI MRI slice, or (3) 1 representative T2FS MRI slice. All 3 grading systems were used.</p></list-item><list-item><p>ID: Simulating the workflow of clinical surgeons, the model is provided with images and corresponding image descriptions to verify whether multimodal text-image correspondence prompts can enhance model performance. Prior work suggests that combining images with corresponding textual descriptions can enhance cross-modal alignment and support multimodal reasoning in vision-language models [<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref27">27</xref>]. The models received the same single images as the SI (X-ray, T1WI, or T2FS), accompanied by a corresponding &#x201C;radiology image description&#x201D; text. The radiology descriptions used in the ID configuration were extracted from the objective findings section of routine clinical radiology reports generated by on-duty reporting and supervising radiologists during daily clinical practice; they were not created specifically for this study. The radiologists who participated in image screening and reference-standard establishment did not author the clinical reports used for the ID prompt descriptions. During data extraction, patient identifiers, diagnostic names, staging information, and conclusion-based diagnostic statements were removed, leaving only descriptions of visible imaging findings. The reference standard was established independently, and the assessors were blinded to the final diagnostic and staging labels contained in the original clinical reports.</p></list-item><list-item><p>MI: This configuration evaluates the model&#x2019;s multiview integration capabilities and simulates the real-world clinical requirement of synthesizing information across multiple images. We aimed to test whether MLLMs could correlate features across different images to improve grading accuracy, specifically using the ARCO system. For each case, all MI images were submitted simultaneously within a single model call using the same ARCO grading prompt. Images were organized in a fixed order for consistency, but no explicit fusion strategy or method was imposed. No additional investigator-defined maximum token limit was set beyond the default constraints of each model interface. Two setups were tested:</p><list list-type="bullet"><list-item><p>Multi-image setup 1 (MIS1): All 4 models were tested. The inputs included 1 anteroposterior pelvic radiograph followed by 4 T1WI MRI slices and 4 T2FS MRI slices from the same patient.</p></list-item><list-item><p>Multi-image setup 2 (MIS2): It evaluated models (Claude and Gemma) that performed well in the SI-MRI tests. The inputs included 4 representative T1WI MRI slices, followed by 4 T2FS MRI slices from the same patient.</p></list-item></list></list-item></list><p>To ensure comparability across models, all MLLMs were evaluated using the same case order, image inputs, prompt templates, task instructions, and output format requirements within each prompt configuration. All models received the same images, which were exported directly from the institutional PACS in JPG format. Apart from deidentification, no image enhancement, cropping, or other postprocessing was performed before model evaluation in order to approximate the image-input conditions of routine clinical practice.</p><p>We provided the surgeons with the same images and corresponding radiological descriptions used in the ID configuration to simulate routine clinical practice. Therefore, the comparison between surgeons and MLLMs under the ID configuration was based on matched input information. The SI configuration was designed to assess zero-shot image interpretation, whereas the MI configuration was designed to explore MI reasoning.</p></sec><sec id="s2-8"><title>Performance Evaluation</title><sec id="s2-8-1"><title>Diagnostic Performance of ONFH Detection</title><p>ONFH detection was evaluated based on each model&#x2019;s ability to diagnose the presence or absence of ONFH, using confidence scores ranging from 0 to 1 for each hip. The primary metric was the area under the receiver operating characteristic curve (AUC). Binary classification metrics, including accuracy, sensitivity, specificity, precision, recall, and <italic>F</italic><sub>1</sub>-score, were calculated using a prespecified confidence-score threshold of 0.5. Model outputs with confidence scores of 0.5 or higher were classified as positive for ONFH, whereas scores below 0.5 were classified as negative.</p><p>Accuracy = (TP + TN)/(TP + TN + FP + FN)</p><p>Sensitivity = TP/(TP + FN)</p><p>Specificity = TN/(TN + FP)</p><p>Where TP is true positive, TN is true negative, FP is false positive, and FN is false negative.</p></sec><sec id="s2-8-2"><title>Performance of Early- or Late-Stage ONFH Differentiation</title><p>Performance in early versus late ONFH staging was assessed. Reference grades were categorized as &#x201C;early&#x201D; (Ficat 1&#x2010;2; ARCO 0&#x2010;2; Steinberg 0&#x2010;3) or &#x201C;late&#x201D; (Ficat 3&#x2010;4; ARCO 3&#x2010;4; Steinberg 4&#x2010;6). Precision, recall, <italic>F</italic><sub>1</sub>-score, and overall accuracy were calculated from these categorical predictions; no additional probability threshold was applied.</p><p>Precision = TP/(TP + FP)</p><p>Recall = TP/(TP + FN)</p><p><italic>F</italic><sub>1</sub>-score = 2 &#x00D7; (precision &#x00D7; recall)/(precision + recall)</p></sec><sec id="s2-8-3"><title>Detailed Grading Performance</title><p>Detailed grading performance, including subgrades, was assessed using the Ficat, ARCO, and Steinberg systems. Key metrics included overall grading accuracy (the proportion of grades matching the reference). Confusion matrices illustrated classification patterns and errors.</p></sec></sec><sec id="s2-9"><title>Reliability Assessment</title><p>Grading reliability was assessed using ICCs. Surgeon interrater reliability was calculated from the independent grading results of three orthopedic surgeons. Surgeon intrarater reliability was assessed by asking the surgeons to regrade a randomized subset of images after a 2-week interval. MLLM output stability was assessed by comparing 2 independent runs on the same dataset for each input configuration (SI, ID, and MI). Reliability analysis was limited to ARCO grading because ARCO was the only grading framework that was consistently applied across all prompt configurations, imaging cohorts, MLLM models, and clinical readers, including the MI settings. This approach allowed for a comparable run-to-run repeatability assessment for MLLMs and agreement analysis with physician grading across the full study design. ICCs were calculated using a 2-way mixed-effects model with absolute agreement, where values &#x003E;0.75 were excellent, 0.5 to 0.75 fair to good, and &#x003C;0.5 poor, with significance at <italic>P</italic>&#x003C;.05 [<xref ref-type="bibr" rid="ref28">28</xref>].</p></sec><sec id="s2-10"><title>Statistical Analysis</title><p>Statistical analyses were performed using SPSS v26.0 (IBM Corp) and custom statistical scripts for clustered resampling analyses. The individual hip within each imaging modality was retained as the primary unit of analysis because disease status and stage may differ between sides. Statistical significance was defined as a 2-sided <italic>P</italic>&#x003C;.05.</p><p>The sample size calculation was performed for the overall cohort and indicated that 288 hips were required, assuming an expected accuracy of 75%, a 95% confidence level, and a 5% margin of error. The final cohort included 658 modality-specific hip observations, comprising 318 radiographic and 340 MRI observations, exceeding the overall sample size requirement. No separate sample size calculations were performed for modality-specific subgroups or individual disease-stage categories.</p><p>Descriptive statistics were used to summarize cohort characteristics. To account for within-patient correlations arising from bilateral hips and from patients contributing observations to more than 1 imaging modality, generalized estimating equations with patient ID as the clustering variable were used for binary correctness outcomes. For AUC, sensitivity, specificity, <italic>F</italic><sub>1</sub>-score, staging accuracy, and ICC, 95% CIs were estimated using patient-clustered bootstrap resampling with 2000 repetitions, in which patients rather than individual hip observations were resampled, and all hip-level observations from selected patients were retained. Paired patient-clustered bootstrap analysis was used to compare SI and ID configurations. A one-hip-per-patient sensitivity analysis was also conducted to assess robustness. When multiple pairwise comparisons were performed within the same analysis family, Bonferroni correction was applied to control for multiple testing.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Baseline Cohort Characteristics</title><p>The patient selection and cohort construction process is shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>. The patient cohort included 159 radiographic patients contributing 318 hip-level observations and 170 MRI patients contributing 340 hip-level observations; 55 patients with both modalities formed the MI subgroup (<xref ref-type="fig" rid="figure2">Figure 2</xref>). Radiographic patients (mean age 59.2, SD 14.5 y; 81 male patients, 78 female patients) had a higher proportion of late-stage ONFH (119/318 hips, 37%), while MRI patients (mean age 54.5, SD 17.2 y; 101 male patients, 69 female patients) had more early-stage disease (278/340 hips, 82%; <xref ref-type="table" rid="table1">Table 1</xref>).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Data collection flowchart. MRI: magnetic resonance imaging; ONFH: osteonecrosis of the femoral head.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92919_fig02.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Distribution of osteonecrosis of the femoral head (ONFH) grades (per-hip) in the study cohort<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Grade</td><td align="left" valign="bottom" colspan="3">Radiograph</td><td align="left" valign="bottom" colspan="3">MRI<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> image</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Ficat, n (%)</td><td align="left" valign="bottom">ARCO<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>, n (%)</td><td align="left" valign="bottom">Steinberg, n (%)</td><td align="left" valign="bottom">Ficat, n (%)</td><td align="left" valign="bottom">ARCO, n (%)</td><td align="left" valign="bottom">Steinberg, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">0</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">135 (42.6)</td><td align="left" valign="top">125 (39.3)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">91 (26.8)</td><td align="left" valign="top">120 (35.3)</td></tr><tr><td align="left" valign="top">1</td><td align="left" valign="top">135 (42.6)</td><td align="left" valign="top">0 (0.0)</td><td align="left" valign="top">3 (0.9)</td><td align="left" valign="top">133 (39.1)</td><td align="left" valign="top">61 (17.9)</td><td align="left" valign="top">39 (11.5)</td></tr><tr><td align="left" valign="top">2</td><td align="left" valign="top">76 (23.9)</td><td align="left" valign="top">63 (19.8)</td><td align="left" valign="top">68 (21.4)</td><td align="left" valign="top">145 (42.6)</td><td align="left" valign="top">109 (32.1)</td><td align="left" valign="top">100 (29.4)</td></tr><tr><td align="left" valign="top">3</td><td align="left" valign="top">37 (11.6)</td><td align="left" valign="top">43 (13.5)</td><td align="left" valign="top">4 (1.3)</td><td align="left" valign="top">26 (7.6)</td><td align="left" valign="top">45 (13.2)</td><td align="left" valign="top">16 (4.7)</td></tr><tr><td align="left" valign="top">4</td><td align="left" valign="top">70 (22.0)</td><td align="left" valign="top">76 (24.0)</td><td align="left" valign="top">41 (12.9)</td><td align="left" valign="top">36 (10.6)</td><td align="left" valign="top">34 (10)</td><td align="left" valign="top">34 (10)</td></tr><tr><td align="left" valign="top">5</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">46 (14.5)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">22 (6.5)</td></tr><tr><td align="left" valign="top">6</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">31 (9.7)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">9 (2.6)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Distribution of osteonecrosis of the femoral head grades according to the Ficat, Association Research Circulation Osseous, and Steinberg systems for the radiographic and magnetic resonance imaging cohorts.</p></fn><fn id="table1fn2"><p><sup>b</sup>MRI: magnetic resonance imaging.</p></fn><fn id="table1fn3"><p><sup>c</sup>ARCO: Association Research Circulation Osseous.</p></fn><fn id="table1fn4"><p><sup>d</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Diagnostic Performance of ONFH Detection</title><p>ONFH diagnosis by MLLMs strongly depended on prompt configuration (<xref ref-type="fig" rid="figure3">Figure 3</xref>). Surgeons showed high diagnostic accuracy, correlating with experience (A&#x003E;B&#x003E;C; <italic>P</italic>&#x003C;.001). With SI input, MLLMs showed limited diagnostic discrimination, with AUCs ranging from 0.50 to 0.58 and a mean AUC of 0.55 (SD 0.03). ID input substantially improved detection performance, with AUCs ranging from 0.90 to 0.93 and a mean AUC of 0.91 (SD 0.01). Patient-clustered bootstrap analyses showed that ID significantly increased detection AUC compared with SI for all 4 models in both radiograph and MRI cohorts (all <italic>P</italic>&#x003C;.001). Generalized estimating equation (GEE) analyses similarly showed significantly higher diagnostic correctness with ID than with SI for every model in both imaging cohorts (all <italic>P</italic>&#x003C;.001). In contrast, MI input showed limited and inconsistent diagnostic discrimination, with a mean AUC of 0.53 (SD 0.08). The one-hip-per-patient sensitivity analysis yielded similar results, with AUCs of 0.52 (95% CI 0.50&#x2010;0.54) for SI, 0.92 (95% CI 0.91&#x2010;0.93) for ID, and 0.55 (95% CI 0.47&#x2010;0.63) for MI. Model-specific AUC estimates and 95% CIs are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Diagnostic performance based on receiver operating characteristic (ROC) curves and binary classification metrics. (A) ROC curves for surgeons and multimodal large language models (MLLMs) with radiographic inputs (single image [SI], image plus radiology description [ID]). (B) ROC curves for surgeons and MLLMs with magnetic resonance imaging (MRI) inputs (SI, ID). (C) ROC curves for surgeons and MLLMs in the multi-image (MI) configuration (multi-image setup 1 [MIS1], multi-image setup 2 [MIS2]). (D) Performance metrics at a 0.5 threshold. AUC: area under the receiver operating characteristic curve.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92919_fig03.png"/></fig></sec><sec id="s3-3"><title>Performance of Early- or Late-Stage ONFH Differentiation</title><p>For early- and late-stage ONFH differentiation, MLLM performance was also strongly influenced by prompt configuration, with early-stage detection being better than late-stage detection (<xref ref-type="fig" rid="figure4">Figure 4</xref>). Across the Ficat, ARCO, and Steinberg systems, the mean early- or late-stage accuracy increased from approximately 0.65 (SD 0.11) with SI input to 0.78 (SD 0.04) with ID input. For ARCO-based early or late differentiation, ID significantly improved accuracy over SI for all 4 models in the radiograph cohort (all <italic>P</italic>&#x003C;.001). In the MRI cohort, ID significantly improved ARCO early or late accuracy for Qwen and ChatGPT (both <italic>P</italic>&#x003C;.001) and Gemma (<italic>P</italic>=.01), whereas no significant improvement was observed for Claude (<italic>P</italic>=.50). MI input did not show a consistent advantage, with the mean ARCO early or late accuracy of approximately 0.59 (SD 0.10).</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Early- and late-stage osteonecrosis of the femoral head (ONFH) differentiation and detailed grading accuracy. (A) Performance metrics (precision, recall, <italic>F</italic><sub>1</sub>-score) for early ONFH differentiation (single image [SI], image plus radiology description [ID]). (B) Performance metrics for early ONFH differentiation (multi-image setup 1 [MIS1], multi-image setup 2 [MIS2]). (C) Performance metrics for late ONFH differentiation (SI, ID). (D) Performance metrics for late ONFH differentiation (MIS1, MIS2). (E) Accuracy of early- and late-stage ONFH differentiation (SI, ID). (F) Accuracy for early and late differentiation (MIS1, MIS2). (G) The detailed accuracy of surgeons and multimodal large language models (MLLMs) in grading ONFH differentiation in the SI and ID configurations. (H) The detailed accuracy of surgeons and MLLMs in grading ONFH differentiation in the MIS1 and MIS2 configurations. ARCO: Association Research Circulation Osseous; MI: multi-image; MRI: magnetic resonance imaging.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92919_fig04.png"/></fig></sec><sec id="s3-4"><title>Detailed Grading Performance</title><p>Detailed grading performance remained more challenging than binary ONFH detection. Surgeons achieved a mean accuracy of 0.71 (SD 0.19). Nevertheless, ID input consistently improved grading accuracy compared with SI input across the 3 grading systems (<xref ref-type="fig" rid="figure4">Figure 4G,H</xref>). The mean detailed grading accuracy increased from 0.34 (SD 0.08) to 0.59 (SD 0.04) for Ficat, from 0.22 (SD 0.05) to 0.56 (SD 0.11) for ARCO, and from 0.19 (SD 0.05) to 0.49 (SD 0.05) for Steinberg. Patient-clustered bootstrap analyses showed that ID significantly outperformed SI for all models in both radiograph and MRI cohorts (<italic>P</italic>&#x003C;.001). MI input provided limited detailed ARCO grading performance, with a mean accuracy of approximately 0.24 (SD 0.03).</p></sec><sec id="s3-5"><title>Performance Comparison of Different MLLMs</title><p>This study compared the performance of different MLLMs, distinguishing between commercial and open-source solutions. Consistent with the bootstrap analysis, no significant overall difference was found between commercial and open-source models, though some model-specific variations existed (<xref ref-type="fig" rid="figure5">Figure 5C,D</xref>). Furthermore, individual models exhibited notable heterogeneity in diagnostic strategies for ONFH detection. Claude and Gemma demonstrated high sensitivity (mean 0.81, SD 0.13) but low specificity (mean 0.30, SD 0.13), suggesting screening use, while Qwen adopted a high-specificity (mean 0.85, SD 0.14), low-sensitivity (mean 0.20, SD 0.11) approach suited for confirmatory roles; ChatGPT showed a balanced sensitivity and specificity. This model-specific behavior was modality-contingent, with open-source Gemma showing higher accuracy than commercial ChatGPT in MRI-based differentiation, but ChatGPT showing higher accuracy with radiographs (<xref ref-type="fig" rid="figure5">Figure 5C,E</xref>).</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Overall multimodal large language model (MLLM) performance analysis. (A) Confusion matrix of surgeons, ChatGPT, and Qwen in the Association Research Circulation Osseous (ARCO) grading system under the magnetic resonance imaging (MRI) single image (SI) and image plus radiology description (ID) configurations. (B) Overall accuracy for surgeons and MLLMs across the SI, ID, multi-image setup 1 (MIS1), and multi-image setup 2 (MIS2) configurations. (C) Comparison of overall accuracy between the radiographs and MRI images. (D) Comparison of overall accuracy between commercial and open-source MLLMs. (E) Overall accuracy of individual MLLMs. (F) Comparison of overall accuracy across multimodal prompt configurations. Bar plots show descriptive overall accuracy across models, imaging inputs, and prompt configurations. Formal patient-clustered statistical comparisons between SI, ID, and MI configurations are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92919_fig05.png"/></fig></sec><sec id="s3-6"><title>Impact of Multimodal Prompt Configurations</title><p>Prompt configuration had a marked effect on model performance (<xref ref-type="fig" rid="figure5">Figure 5</xref>). Across all models and imaging modalities, the mean detection AUC increased from 0.55 (SD 0.03) with SI input to 0.91 (SD 0.01) with ID input, corresponding to an average AUC gain of 0.37 (SD 0.03). The AUC improvement was significant for each model in both radiograph and MRI cohorts after patient-clustered bootstrap correction (all <italic>P</italic>&#x003C;.001). ID input also improved ARCO detailed grading accuracy from 0.22 to 0.56 and ARCO early- or late-stage accuracy from 0.59 to 0.79. By contrast, MI input did not produce consistent performance gains over SI input. In paired patient-clustered bootstrap comparisons, MI AUCs were similar to SI AUCs, with mean AUC differences of &#x2212;0.03 (SD 0.07) for MIS1 versus radiograph SI, &#x2212;0.05 (SD 0.10)for MIS1 versus MRI SI, and &#x2212;0.002 (SD 0.03) for MIS2 versus MRI SI; most model- and modality-specific comparisons were not significant. In contrast, MI AUCs were significantly lower than the corresponding ID AUCs, with mean AUC differences of &#x2212;0.38 (SD 0.09) for MIS1 versus radiograph ID, &#x2212;0.40 (SD 0.10) for MIS1 versus MRI ID, and &#x2212;0.37 (SD 0.05) for MIS2 versus MRI ID across all comparisons (all <italic>P</italic>&#x003C;.001). These findings suggest that simply increasing the number of images did not reliably improve current MLLM reasoning in this task. Detailed paired bootstrap and GEE results are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s3-7"><title>Reliability of Grading</title><p>To assess grading reliability, we evaluated ARCO grading consistency using ICCs (<xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>). Reliability analysis was limited to ARCO because it was the only grading framework applied across all prompt configurations, including MI. Human readers showed moderate-to-good interrater agreement, with ICCs of 0.74 for radiographs and 0.72 for MRI images. In the SI configuration, MLLM repeatability varied substantially across models and modalities, with a mean ICC of 0.51 (SD 0.18). ID input markedly improved model repeatability, with a mean ICC of 0.97 (SD 0.02) and consistently high ICCs across models. In contrast, MI configurations showed lower and less stable reliability, with mean ICCs of 0.52 (SD 0.18) for MIS1 and 0.43 (SD 0.01) for MIS2.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Intraclass correlation coefficients for multimodal large language model Association Research Circulation Osseous grading reliability<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Imaging input</td><td align="left" valign="bottom">Prompt configuration</td><td align="left" valign="bottom">ChatGPT</td><td align="left" valign="bottom">Claude</td><td align="left" valign="bottom">Qwen</td><td align="left" valign="bottom">Gemma</td></tr></thead><tbody><tr><td align="left" valign="top">Radiographs</td><td align="left" valign="top">SI<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">0.76</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.52</td><td align="left" valign="top">0.29</td></tr><tr><td align="left" valign="top">Radiographs</td><td align="left" valign="top">ID<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">0.97</td><td align="left" valign="top">0.98</td><td align="left" valign="top">0.96</td><td align="left" valign="top">0.97</td></tr><tr><td align="left" valign="top">MRI<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup> images</td><td align="left" valign="top">SI</td><td align="left" valign="top">0.38</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.37</td><td align="left" valign="top">0.44</td></tr><tr><td align="left" valign="top">MRI images</td><td align="left" valign="top">ID</td><td align="left" valign="top">0.95</td><td align="left" valign="top">0.99</td><td align="left" valign="top">0.96</td><td align="left" valign="top">0.93</td></tr><tr><td align="left" valign="top">Multi-images</td><td align="left" valign="top">MIS1<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top">0.37</td><td align="left" valign="top">0.75</td><td align="left" valign="top">0.37</td><td align="left" valign="top">0.59</td></tr><tr><td align="left" valign="top">Multi-images</td><td align="left" valign="top">MIS2<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup></td><td align="left" valign="top">0.43</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.45</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Intraclass correlation coefficients indicate intramodel repeatability for Association Research Circulation Osseous grading across two independent runs.</p></fn><fn id="table2fn2"><p><sup>b</sup>SI: single image.</p></fn><fn id="table2fn3"><p><sup>c</sup>ID: image plus radiology description.</p></fn><fn id="table2fn4"><p><sup>d</sup>MRI: magnetic resonance imaging.</p></fn><fn id="table2fn5"><p><sup>e</sup>MIS1: multi-image setup 1.</p></fn><fn id="table2fn6"><p><sup>f</sup>MIS2: multi-image setup 2.</p></fn><fn id="table2fn7"><p><sup>g</sup>The em dash indicates that the corresponding model-configuration combination was not evaluated.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Intraclass correlation coefficients for surgeon Association Research Circulation Osseous grading reliability<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Reference surgeon</td><td align="left" valign="top">Imaging input</td><td align="left" valign="top">Surgeon A</td><td align="left" valign="top">Surgeon B</td><td align="left" valign="top">Surgeon C</td></tr></thead><tbody><tr><td align="left" valign="top">Surgeon A</td><td align="left" valign="top">Radiographs</td><td align="char" char="." valign="top">0.92</td><td align="char" char="." valign="top">0.87</td><td align="char" char="." valign="top">0.69</td></tr><tr><td align="left" valign="top">Surgeon A</td><td align="left" valign="top">MRI<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> images</td><td align="char" char="." valign="top">0.81</td><td align="char" char="." valign="top">0.79</td><td align="char" char="." valign="top">0.73</td></tr><tr><td align="left" valign="top">Surgeon B</td><td align="left" valign="top">Radiographs</td><td align="char" char="." valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="char" char="." valign="top">0.87</td><td align="char" char="." valign="top">0.67</td></tr><tr><td align="left" valign="top">Surgeon B</td><td align="left" valign="top">MRI images</td><td align="left" valign="top">&#x2014;</td><td align="char" char="." valign="top">0.89</td><td align="char" char="." valign="top">0.61</td></tr><tr><td align="left" valign="top">Surgeon C</td><td align="left" valign="top">Radiographs</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="char" char="." valign="top">0.72</td></tr><tr><td align="left" valign="top">Surgeon C</td><td align="left" valign="top">MRI images</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="char" char="." valign="top">0.78</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Intraclass correlation coefficients indicate surgeon Association Research Circulation Osseous grading reliability. Diagonal values indicate intrareader reliability, and off-diagonal values indicate pairwise interreader reliability.</p></fn><fn id="table3fn2"><p><sup>b</sup>MRI: magnetic resonance imaging.</p></fn><fn id="table3fn3"><p><sup>c</sup>The em dash indicates duplicate pairwise comparisons that are not repeated in the table.</p></fn></table-wrap-foot></table-wrap><p>To better characterize run-to-run variability, we additionally calculated exact agreement, within-one-stage agreement, and mean absolute ARCO stage difference. ID input showed the highest stability, with exact agreement of 0.79&#x2010;0.96 (mean 0.90, SD 0.05), within-one-stage agreement of 0.98&#x2010;1.00 (mean 0.99, SD 0.01), and mean absolute differences of 0.05&#x2010;0.23 stages (mean 0.11, SD 0.06). In contrast, SI input showed lower stability, with exact agreement of 0.44&#x2010;0.81 (mean 0.65, SD 0.16) and mean absolute differences of 0.22&#x2010;1.09 stages (mean 0.60, SD 0.27). MI configurations showed inconsistent repeatability, with exact agreement of 0.47&#x2010;0.85 (mean 0.67, SD 0.14) and mean absolute differences of 0.18&#x2010;0.74 stages (mean 0.44, SD 0.23).</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study evaluated whether MLLMs can assist in ONFH diagnosis and staging, whether performance differs across commercial and open-source models, and whether prompt configuration affects diagnostic reliability. The main findings were that prompt configuration strongly influenced MLLM performance, ID input substantially improved diagnostic and grading performance, MI input did not provide consistent additional benefit, and different models showed task-specific strengths without a consistent advantage of commercial over open-source models.</p></sec><sec id="s4-2"><title>Interpretation and Comparison With Previous Work</title><p>These findings should be interpreted in the context of previous work on medical imaging AI and multimodal models. In terms of diagnostic and staging performance, the evaluated MLLMs demonstrated diagnostic performance competitive with existing task-specific DL models in the ID configuration (Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). For example, while prior studies reported an AUC of 0.80 for stage differentiation [<xref ref-type="bibr" rid="ref29">29</xref>] and 88% accuracy for diagnosis [<xref ref-type="bibr" rid="ref15">15</xref>] using DL, MLLMs in the ID configuration achieved a mean early/late classification accuracy of approximately 0.78 (SD 0.04) and a mean detection AUC of 0.91 (SD 0.01) for ONFH diagnosis. However, these values were obtained from different datasets, patient populations, study designs, and evaluation procedures and should not be interpreted as a direct performance comparison. By contrast, the SI configuration represented zero-shot image-only interpretation and showed substantially lower performance, underscoring the current limitations of autonomous MLLM-based diagnosis. Unlike task-specific DL models trained on labeled datasets, the MLLMs in this study were evaluated without task-specific training or fine-tuning. The comparison presented here serves only as a reference for diagnostic performance. These findings demonstrate the potential of MLLMs to support human-AI collaborative workflows in orthopedics. Traditional DL models are typically trained for single-label tasks, which limits their adaptability. In contrast, MLLMs can integrate both image and textual inputs without task-specific fine-tuning, enabling them to generalize across different grading systems and perform multiple task types. This multimodal capability and strong generalization ability demonstrate their promise for broader clinical application translation.</p><p>Most prior research has focused on commercial models [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>] without systematically comparing the performance of different MLLMs in clinical tasks. However, this research has found that different models displayed modality-specific strengths (eg, Gemma &#x003E; ChatGPT in MRI; Qwen &#x003E; Claude in radiographs) and nuanced diagnostic strategies (eg, Qwen: high specificity for confirmation; Claude/Gemma: high sensitivity for screening). These findings have rarely been mentioned in previous comparative studies [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. These observations suggest that different MLLMs may exhibit distinct performance tendencies under specific imaging conditions. Models with higher sensitivity may be more suitable for screening, where missed early ONFH could delay joint-preserving treatment, whereas models with higher specificity may be more appropriate for confirmatory support to reduce unnecessary examinations caused by false-positive results. Nevertheless, none of the evaluated models should currently be used independently for clinical decision-making.</p><p>From a clinical perspective, these results suggest that different MLLMs may be suited to different assistive roles rather than a single universal diagnostic use case. Furthermore, our research, along with other recent studies, indicates that open-source alternatives hold significant potential [<xref ref-type="bibr" rid="ref34">34</xref>]. From a deployment perspective, commercial models offer easy access via cloud-based APIs but may involve ongoing costs, network latency, vendor dependence, and privacy, security, and data-governance issues. In contrast, open-source models support on-premises deployment, institutional customization, and version control and can better protect sensitive clinical data [<xref ref-type="bibr" rid="ref35">35</xref>]. As this study focused on diagnostic performance, inference time, computational costs, and deployment requirements were not systematically measured; these aspects should be evaluated in future research.</p><p>The significant impact of multimodal prompt configuration design on model performance is well documented [<xref ref-type="bibr" rid="ref36">36</xref>], and our study confirms that multimodal prompting enhances all MLLMs&#x2019; performance by radiology-description prompting. ID prompts, which combine images with radiologist-generated descriptions (extracted from original clinical reports without diagnostic conclusion information), significantly improve diagnostic accuracy and grading precision for both X-ray and MRI modalities. The accompanying radiology descriptions may improve performance by directing model attention toward clinically relevant imaging findings and facilitating cross-modal alignment [<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref27">27</xref>]. Notably, this approach improves interrater agreement (ICC=0.96) to a level that approached or exceeded physician-reader agreement in selected comparisons [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Counterintuitively, the MI configuration, which mimics clinical workflows, failed to provide substantial benefits. Recent MI benchmarks have shown that although MLLMs perform well in many single-image tasks, they still have limitations in fine-grained perception, cross-image comparison, and MI reasoning [<xref ref-type="bibr" rid="ref37">37</xref>]. Multi-image inputs may also increase the risk of context dilution and unstable attention allocation, where subtle ONFH findings on a key slice may be underweighted when multiple radiographs, T1WI, and T2FS images are submitted together. This explanation is consistent with recent work showing that hallucination and reasoning instability can increase in MI settings and may be influenced by interimage attention distribution [<xref ref-type="bibr" rid="ref38">38</xref>]. The results suggest that integrating radiologist-generated image descriptions not only increases diagnostic performance but also enhances model stability. This suggests that the practical value of MLLMs may currently lie in integrating clinician-generated imaging descriptions with visual inputs within human-AI collaborative workflows, underscoring the practical value of human-AI collaboration in real-world clinical applications [<xref ref-type="bibr" rid="ref39">39</xref>].</p><p>These findings suggest that MLLMs may have practical value as assistive tools within human-AI collaborative workflows in orthopedic imaging. When radiologist-generated imaging descriptions were available, MLLMs showed improved diagnostic and grading performance in selected tasks, supporting their potential role in standardizing ONFH assessment and organizing imaging findings into structured outputs. The comparable performance of open-source and commercial models also suggests that institutionally controlled and lower-cost deployment strategies may be feasible. However, these results should be interpreted as preliminary, and prospective multicenter studies are needed to evaluate workflow efficiency, clinical safety, and real-world use before implementation.</p></sec><sec id="s4-3"><title>Limitations</title><p>This study has several limitations. First, the single-center dataset limits generalizability. Differences across institutions in imaging protocols, disease prevalence, patient demographics, and annotation practices may affect model performance. External validation using multicenter datasets is therefore required before clinical application. Second, the MI analysis was exploratory; the MI cohort was relatively small, and we did not investigate the mechanisms underlying the underperformance of MI configurations. Third, the MRI subgroup contained a higher proportion of early-stage ONFH cases, and this class imbalance may have affected stage-specific performance estimates. Fourth, this study did not include a text-only control condition using radiology descriptions without images. Therefore, the relative contribution of textual radiology descriptions and image inputs to the improved ID performance could not be quantified. Fifth, repeatability was assessed mainly for ARCO grading, and although ICC and run-to-run variability metrics were reported, no established threshold currently defines clinically acceptable MLLM variability for ONFH staging. Moreover, deterministic generation settings could not be fully ensured across all models, which may have contributed to output variability. Finally, because bone scintigraphy was unavailable, ARCO stage 0 was operationally defined in this study as the absence of radiographic abnormalities, and ARCO-related results should therefore be interpreted within this radiograph- or MRI-based operational framework. Despite these limitations, our findings offer valuable insights into the potential of MLLMs in medical applications.</p></sec><sec id="s4-4"><title>Conclusions</title><p>This study highlights the importance of input design when applying MLLMs to medical imaging tasks. The broader implication is that MLLMs may be most useful not as stand-alone diagnostic systems but as assistive tools that help standardize imaging interpretation and integrate visual and textual clinical information within human-AI workflows. Open-source models may further support accessible and institutionally controlled deployment in musculoskeletal imaging. However, current MLLMs remain insufficient for independent diagnostic use, and future studies should focus on prospective validation, multicenter testing, robust MI reasoning, and workflow-level evaluation before clinical implementation.</p></sec></sec></body><back><ack><p>Generative AI tools were used to assist with language editing, grammar correction, and organization of selected revisions. They were not used for study design, data collection, diagnostic labeling, reference-standard establishment, statistical analysis, figure or table generation, or scientific interpretation. All AI-assisted content was reviewed and verified by the authors.</p></ack><notes><sec><title>Funding</title><p>This research was supported by the General Program of the Zhejiang Provincial Department of Education (Y202560061). The funder had no role in the study design, data collection, data analysis, data interpretation, manuscript preparation, or decision to submit the manuscript for publication.</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed during this study are not publicly available due to privacy and ethical restrictions related to clinical imaging data. Reasonable requests for deidentified data may be directed to the corresponding author and will be considered subject to institutional approval.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: PF</p><p>Data analysis: JS, SC, PX, ZG</p><p>Data curation: DC</p><p>Formal analysis: PF</p><p>Methodology: PF, JZ</p><p>Project administration: JZ</p><p>Resources: XH, Y Lu, Y Lin, XW, TY</p><p>Software: PF, LL</p><p>Supervision: PF</p><p>Visualization: JZ</p><p>Writing-original draft: JZ</p><p>Writing-review &#x0026; editing: PF, JZ</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ARCO</term><def><p>Association Research Circulation Osseous</p></def></def-item><def-item><term id="abb2">AUC</term><def><p>area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb3">DL</term><def><p>deep learning</p></def></def-item><def-item><term id="abb4">FN</term><def><p>false negative</p></def></def-item><def-item><term id="abb5">FP</term><def><p>false positive</p></def></def-item><def-item><term id="abb6">GEE</term><def><p>generalized estimating equation</p></def></def-item><def-item><term id="abb7">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb8">ID</term><def><p>image plus radiology description</p></def></def-item><def-item><term id="abb9">MI</term><def><p>multi-image</p></def></def-item><def-item><term id="abb10">MIS1</term><def><p>multi-image setup 1</p></def></def-item><def-item><term id="abb11">MIS2</term><def><p>multi-image setup 2</p></def></def-item><def-item><term id="abb12">MLLM</term><def><p>multimodal large language model</p></def></def-item><def-item><term id="abb13">MRI</term><def><p>magnetic resonance imaging</p></def></def-item><def-item><term id="abb14">ONFH</term><def><p>osteonecrosis of the femoral head</p></def></def-item><def-item><term id="abb15">PACS</term><def><p>picture archiving and communication system</p></def></def-item><def-item><term id="abb16">SI</term><def><p>single image</p></def></def-item><def-item><term id="abb17">T1WI</term><def><p>T1-weighted imaging</p></def></def-item><def-item><term id="abb18">T2FS</term><def><p>T2-weighted fat-suppressed</p></def></def-item><def-item><term id="abb19">THA</term><def><p>total hip arthroplasty</p></def></def-item><def-item><term id="abb20">TN</term><def><p>true negative</p></def></def-item><def-item><term id="abb21">TP</term><def><p>true positive</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maillefert</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Tavernier</surname><given-names>C</given-names> </name><name name-style="western"><surname>Toubeau</surname><given-names>M</given-names> </name><name name-style="western"><surname>Brunotte</surname><given-names>F</given-names> </name></person-group><article-title>Non-traumatic avascular necrosis of the femoral head</article-title><source>J Bone Joint Surg Am</source><year>1996</year><month>03</month><volume>78</volume><issue>3</issue><fpage>473</fpage><lpage>474</lpage><pub-id pub-id-type="medline">8613457</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Petek</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hannouche</surname><given-names>D</given-names> </name><name name-style="western"><surname>Suva</surname><given-names>D</given-names> </name></person-group><article-title>Osteonecrosis of the femoral head: pathophysiology and current concepts of treatment</article-title><source>EFORT Open Rev</source><year>2019</year><month>03</month><volume>4</volume><issue>3</issue><fpage>85</fpage><lpage>97</lpage><pub-id pub-id-type="doi">10.1302/2058-5241.4.180036</pub-id><pub-id pub-id-type="medline">30993010</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moya-Angeler</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gianakos</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Villa</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Ni</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lane</surname><given-names>JM</given-names> </name></person-group><article-title>Current concepts on osteonecrosis of the femoral head</article-title><source>World J Orthop</source><year>2015</year><month>09</month><day>18</day><volume>6</volume><issue>8</issue><fpage>590</fpage><lpage>601</lpage><pub-id pub-id-type="doi">10.5312/wjo.v6.i8.590</pub-id><pub-id pub-id-type="medline">26396935</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hernigou</surname><given-names>P</given-names> </name><name name-style="western"><surname>Poignard</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nogier</surname><given-names>A</given-names> </name><name name-style="western"><surname>Manicom</surname><given-names>O</given-names> </name></person-group><article-title>Fate of very small asymptomatic stage-I osteonecrotic lesions of the hip</article-title><source>J Bone Joint Surg Am</source><year>2004</year><month>12</month><volume>86</volume><issue>12</issue><fpage>2589</fpage><lpage>2593</lpage><pub-id pub-id-type="doi">10.2106/00004623-200412000-00001</pub-id><pub-id pub-id-type="medline">15590840</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sen</surname><given-names>RK</given-names> </name></person-group><article-title>Management of avascular necrosis of femoral head at pre-collapse stage</article-title><source>Indian J Orthop</source><year>2009</year><month>01</month><volume>43</volume><issue>1</issue><fpage>6</fpage><lpage>16</lpage><pub-id pub-id-type="doi">10.4103/0019-5413.45318</pub-id><pub-id pub-id-type="medline">19753173</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stoica</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Dumitrescu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Popescu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gheonea</surname><given-names>I</given-names> </name><name name-style="western"><surname>Gabor</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bogdan</surname><given-names>N</given-names> </name></person-group><article-title>Imaging of avascular necrosis of femoral head: familiar methods and newer trends</article-title><source>Curr Health Sci J</source><year>2009</year><month>01</month><volume>35</volume><issue>1</issue><fpage>23</fpage><lpage>28</lpage><pub-id pub-id-type="medline">24778812</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mont</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Marulanda</surname><given-names>GA</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>LC</given-names> </name><etal/></person-group><article-title>Systematic analysis of classification systems for osteonecrosis of the femoral head</article-title><source>J Bone Joint Surg Am</source><year>2006</year><month>11</month><volume>88 Suppl 3</volume><fpage>16</fpage><lpage>26</lpage><pub-id pub-id-type="doi">10.2106/JBJS.F.00457</pub-id><pub-id pub-id-type="medline">17079363</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kay</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Lieberman</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Dorey</surname><given-names>FJ</given-names> </name><name name-style="western"><surname>Seeger</surname><given-names>LL</given-names> </name></person-group><article-title>Inter- and intraobserver variation in staging patients with proven avascular necrosis of the hip</article-title><source>Clin Orthop Relat Res</source><year>1994</year><month>10</month><issue>307</issue><fpage>124</fpage><lpage>129</lpage><pub-id pub-id-type="medline">7924024</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Smith</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Meyer</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Connor</surname><given-names>PM</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Hanley</surname><given-names>EN</given-names> </name></person-group><article-title>Interobserver reliability and intraobserver reproducibility of the modified Ficat classification system of osteonecrosis of the femoral head</article-title><source>J Bone Joint Surg Am</source><year>1996</year><month>11</month><volume>78</volume><issue>11</issue><fpage>1702</fpage><lpage>1706</lpage><pub-id pub-id-type="doi">10.2106/00004623-199611000-00010</pub-id><pub-id pub-id-type="medline">8934485</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaczmarczyk</surname><given-names>R</given-names> </name><name name-style="western"><surname>Wilhelm</surname><given-names>TI</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>R</given-names> </name><name name-style="western"><surname>Roos</surname><given-names>J</given-names> </name></person-group><article-title>Evaluating multimodal AI in medical diagnostics</article-title><source>NPJ Digit Med</source><year>2024</year><month>08</month><day>7</day><volume>7</volume><issue>1</issue><fpage>205</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01208-3</pub-id><pub-id pub-id-type="medline">39112822</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schramm</surname><given-names>S</given-names> </name><name name-style="western"><surname>Preis</surname><given-names>S</given-names> </name><name name-style="western"><surname>Metz</surname><given-names>MC</given-names> </name><etal/></person-group><article-title>Impact of multimodal prompt elements on diagnostic performance of GPT-4V in challenging brain MRI cases</article-title><source>Radiology</source><year>2025</year><month>01</month><volume>314</volume><issue>1</issue><fpage>e240689</fpage><pub-id pub-id-type="doi">10.1148/radiol.240689</pub-id><pub-id pub-id-type="medline">39835982</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>D</given-names> </name><etal/></person-group><article-title>High identification and positive-negative discrimination but limited detailed grading accuracy of ChatGPT-4o in knee osteoarthritis radiographs</article-title><source>Knee Surg Sports Traumatol Arthrosc</source><year>2025</year><month>05</month><volume>33</volume><issue>5</issue><fpage>1911</fpage><lpage>1919</lpage><pub-id pub-id-type="doi">10.1002/ksa.12639</pub-id><pub-id pub-id-type="medline">40053915</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mika</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Engstrom</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Polkowski</surname><given-names>GG</given-names> </name><name name-style="western"><surname>Wilson</surname><given-names>JM</given-names> </name></person-group><article-title>Assessing ChatGPT responses to common patient questions regarding total hip arthroplasty</article-title><source>J Bone Joint Surg Am</source><year>2023</year><month>10</month><day>4</day><volume>105</volume><issue>19</issue><fpage>1519</fpage><lpage>1526</lpage><pub-id pub-id-type="doi">10.2106/JBJS.23.00209</pub-id><pub-id pub-id-type="medline">37459402</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hosny</surname><given-names>A</given-names> </name><name name-style="western"><surname>Parmar</surname><given-names>C</given-names> </name><name name-style="western"><surname>Quackenbush</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schwartz</surname><given-names>LH</given-names> </name><name name-style="western"><surname>Aerts</surname><given-names>HJWL</given-names> </name></person-group><article-title>Artificial intelligence in radiology</article-title><source>Nat Rev Cancer</source><year>2018</year><month>08</month><volume>18</volume><issue>8</issue><fpage>500</fpage><lpage>510</lpage><pub-id pub-id-type="doi">10.1038/s41568-018-0016-5</pub-id><pub-id pub-id-type="medline">29777175</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Deep learning approach for diagnosing early osteonecrosis of the femoral head based on magnetic resonance imaging</article-title><source>J Arthroplasty</source><year>2023</year><month>10</month><volume>38</volume><issue>10</issue><fpage>2044</fpage><lpage>2050</lpage><pub-id pub-id-type="doi">10.1016/j.arth.2022.10.003</pub-id><pub-id pub-id-type="medline">36243276</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>B</given-names> </name><etal/></person-group><article-title>A deep learning-based clinical classification system for the differential diagnosis of hip prosthesis failures using radiographs: a multicenter study</article-title><source>J Bone Joint Surg Am</source><year>2025</year><month>06</month><day>18</day><volume>107</volume><issue>16</issue><fpage>1798</fpage><lpage>1809</lpage><pub-id pub-id-type="doi">10.2106/JBJS.24.01601</pub-id><pub-id pub-id-type="medline">40531980</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dorfner</surname><given-names>FJ</given-names> </name><name name-style="western"><surname>J&#x00FC;rgensen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Donle</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Comparing commercial and open-source large language models for labeling chest radiograph reports</article-title><source>Radiology</source><year>2024</year><month>10</month><volume>313</volume><issue>1</issue><fpage>e241139</fpage><pub-id pub-id-type="doi">10.1148/radiol.241139</pub-id><pub-id pub-id-type="medline">39470431</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gertz</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Dratsch</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bunck</surname><given-names>AC</given-names> </name><etal/></person-group><article-title>Potential of GPT-4 for detecting errors in radiology reports: implications for reporting accuracy</article-title><source>Radiology</source><year>2024</year><month>04</month><volume>311</volume><issue>1</issue><fpage>e232714</fpage><pub-id pub-id-type="doi">10.1148/radiol.232714</pub-id><pub-id pub-id-type="medline">38625012</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sandmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Riepenhausen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Plagwitz</surname><given-names>L</given-names> </name><name name-style="western"><surname>Varghese</surname><given-names>J</given-names> </name></person-group><article-title>Systematic analysis of ChatGPT, Google search and Llama 2 for clinical decision support tasks</article-title><source>Nat Commun</source><year>2024</year><month>03</month><day>6</day><volume>15</volume><issue>1</issue><fpage>2050</fpage><pub-id pub-id-type="doi">10.1038/s41467-024-46411-8</pub-id><pub-id pub-id-type="medline">38448475</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Aggarwal</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lopez</surname><given-names>RD</given-names> </name><name name-style="western"><surname>Namdari</surname><given-names>S</given-names> </name></person-group><article-title>Large language models in orthopaedics: definitions, uses, and limitations</article-title><source>J Bone Joint Surg Am</source><year>2024</year><month>08</month><day>7</day><volume>106</volume><issue>15</issue><fpage>1411</fpage><lpage>1418</lpage><pub-id pub-id-type="doi">10.2106/JBJS.23.01417</pub-id><pub-id pub-id-type="medline">38896652</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hallinan</surname><given-names>JTPD</given-names> </name><name name-style="western"><surname>Leow</surname><given-names>NW</given-names> </name><name name-style="western"><surname>Low</surname><given-names>YX</given-names> </name><etal/></person-group><article-title>An institutional large language model for musculoskeletal MRI improves protocol adherence and accuracy</article-title><source>J Bone Joint Surg Am</source><year>2025</year><month>07</month><day>8</day><volume>107</volume><issue>16</issue><fpage>1833</fpage><lpage>1840</lpage><pub-id pub-id-type="doi">10.2106/JBJS.24.01429</pub-id><pub-id pub-id-type="medline">40627696</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmitt-Sody</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kirchhoff</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mayer</surname><given-names>W</given-names> </name><name name-style="western"><surname>Goebel</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jansson</surname><given-names>V</given-names> </name></person-group><article-title>Avascular necrosis of the femoral head: inter- and intraobserver variations of Ficat and ARCO classifications</article-title><source>Int Orthop</source><year>2008</year><volume>32</volume><issue>3</issue><fpage>283</fpage><lpage>287</lpage><pub-id pub-id-type="doi">10.1007/s00264-007-0320-2</pub-id><pub-id pub-id-type="medline">17396260</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yoon</surname><given-names>BH</given-names> </name><name name-style="western"><surname>Mont</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Koo</surname><given-names>KH</given-names> </name><etal/></person-group><article-title>The 2019 revised version of Association Research Circulation Osseous staging system of osteonecrosis of the femoral head</article-title><source>J Arthroplasty</source><year>2020</year><month>04</month><volume>35</volume><issue>4</issue><fpage>933</fpage><lpage>940</lpage><pub-id pub-id-type="doi">10.1016/j.arth.2019.11.029</pub-id><pub-id pub-id-type="medline">31866252</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Steinberg</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Hayken</surname><given-names>GD</given-names> </name><name name-style="western"><surname>Steinberg</surname><given-names>DR</given-names> </name></person-group><article-title>A quantitative system for staging avascular necrosis</article-title><source>J Bone Joint Surg Br</source><year>1995</year><month>01</month><volume>77</volume><issue>1</issue><fpage>34</fpage><lpage>41</lpage><pub-id pub-id-type="medline">7822393</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Alayrac</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Donahue</surname><given-names>J</given-names> </name><name name-style="western"><surname>Luc</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Flamingo: a visual language model for few-shot learning</article-title><access-date>2026-07-02</access-date><conf-name>NIPS&#x2019;22: Proceedings of the 36th International Conference on Neural Information Processing Systems</conf-name><conf-date>Nov 28 to Dec 9, 2022</conf-date><comment><ext-link ext-link-type="uri" xlink:href="http://www.proceedings.com/68431.html">http://www.proceedings.com/68431.html</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Agarwal</surname><given-names>D</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>J</given-names> </name></person-group><article-title>MedCLIP: contrastive learning from unpaired medical images and text</article-title><source>Proc Conf Empir Methods Nat Lang Process</source><year>2022</year><month>12</month><volume>2022</volume><fpage>3876</fpage><lpage>3887</lpage><pub-id pub-id-type="doi">10.18653/v1/2022.emnlp-main.256</pub-id><pub-id pub-id-type="medline">39144675</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>C</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name></person-group><article-title>Visual prior-based cross-modal alignment network for radiology report generation</article-title><source>Comput Biol Med</source><year>2023</year><month>11</month><volume>166</volume><fpage>107522</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.107522</pub-id><pub-id pub-id-type="medline">37820559</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koo</surname><given-names>TK</given-names> </name><name name-style="western"><surname>Li</surname><given-names>MY</given-names> </name></person-group><article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research</article-title><source>J Chiropr Med</source><year>2016</year><month>06</month><volume>15</volume><issue>2</issue><fpage>155</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id><pub-id pub-id-type="medline">27330520</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Klontzas</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Vassalou</surname><given-names>EE</given-names> </name><name name-style="western"><surname>Spanakis</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Deep learning enables the differentiation between early and late stages of hip avascular necrosis</article-title><source>Eur Radiol</source><year>2024</year><month>02</month><volume>34</volume><issue>2</issue><fpage>1179</fpage><lpage>1186</lpage><pub-id pub-id-type="doi">10.1007/s00330-023-10104-5</pub-id><pub-id pub-id-type="medline">37581656</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pradhan</surname><given-names>P</given-names> </name></person-group><article-title>Accuracy of ChatGPT 3.5, 4.0, 4o and Gemini in diagnosing oral potentially malignant lesions based on clinical case reports and image recognition</article-title><source>Med Oral Patol Oral Cir Bucal</source><year>2025</year><month>03</month><day>1</day><volume>30</volume><issue>2</issue><fpage>e224</fpage><lpage>e231</lpage><pub-id pub-id-type="doi">10.4317/medoral.26824</pub-id><pub-id pub-id-type="medline">39864088</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chambara</surname><given-names>N</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Assessing the feasibility of ChatGPT-4o and Claude 3-Opus in thyroid nodule classification based on ultrasound images</article-title><source>Endocrine</source><year>2025</year><month>03</month><volume>87</volume><issue>3</issue><fpage>1041</fpage><lpage>1049</lpage><pub-id pub-id-type="doi">10.1007/s12020-024-04066-x</pub-id><pub-id pub-id-type="medline">39394537</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nguyen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Carrion</surname><given-names>D</given-names> </name><name name-style="western"><surname>Badawy</surname><given-names>MK</given-names> </name></person-group><article-title>Comparative performance of Anthropic Claude and OpenAI GPT models in basic radiological imaging tasks</article-title><source>J Med Imaging Radiat Oncol</source><year>2025</year><month>06</month><volume>69</volume><issue>4</issue><fpage>431</fpage><lpage>439</lpage><pub-id pub-id-type="doi">10.1111/1754-9485.13858</pub-id><pub-id pub-id-type="medline">40196917</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Adams</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Truhn</surname><given-names>D</given-names> </name><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Llama 3 challenges proprietary state-of-the-art large language models in radiology board-style examination questions</article-title><source>Radiology</source><year>2024</year><month>08</month><volume>312</volume><issue>2</issue><fpage>e241191</fpage><pub-id pub-id-type="doi">10.1148/radiol.241191</pub-id><pub-id pub-id-type="medline">39136566</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nowak</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wulff</surname><given-names>B</given-names> </name><name name-style="western"><surname>Layer</surname><given-names>YC</given-names> </name><etal/></person-group><article-title>Privacy-ensuring open-weights large language models are competitive with closed-weights GPT-4o in extracting chest radiography findings from free-text reports</article-title><source>Radiology</source><year>2025</year><month>01</month><volume>314</volume><issue>1</issue><fpage>e240895</fpage><pub-id pub-id-type="doi">10.1148/radiol.240895</pub-id><pub-id pub-id-type="medline">39807977</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dennst&#x00E4;dt</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hastings</surname><given-names>J</given-names> </name><name name-style="western"><surname>Putora</surname><given-names>PM</given-names> </name><name name-style="western"><surname>Schmerder</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cihoric</surname><given-names>N</given-names> </name></person-group><article-title>Implementing large language models in healthcare while balancing control, collaboration, costs and security</article-title><source>NPJ Digit Med</source><year>2025</year><month>03</month><day>6</day><volume>8</volume><issue>1</issue><fpage>143</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01476-7</pub-id><pub-id pub-id-type="medline">40050366</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Savage</surname><given-names>T</given-names> </name><name name-style="western"><surname>Nayak</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>R</given-names> </name><name name-style="western"><surname>Rangan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>JH</given-names> </name></person-group><article-title>Diagnostic reasoning prompts reveal the potential for large language model interpretability in medicine</article-title><source>NPJ Digit Med</source><year>2024</year><month>01</month><day>24</day><volume>7</volume><issue>1</issue><fpage>20</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01010-1</pub-id><pub-id pub-id-type="medline">38267608</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name><etal/></person-group><article-title>MIBench: evaluating multimodal large language models over multiple images</article-title><conf-name>Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 12-16, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.1250</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>MIHBench: benchmarking and mitigating multi-image hallucinations in multimodal large language models</article-title><conf-name>MM &#x2019;25: Proceedings of the 33rd ACM International Conference on Multimedia</conf-name><conf-date>Oct 27-31, 2025</conf-date><conf-loc>Dublin, Ireland</conf-loc><fpage>3143</fpage><lpage>3152</lpage><pub-id pub-id-type="doi">10.1145/3746027.3754993</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kwong</surname><given-names>JCC</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>SCY</given-names> </name><name name-style="western"><surname>Nickel</surname><given-names>GC</given-names> </name><name name-style="western"><surname>Cacciamani</surname><given-names>GE</given-names> </name><name name-style="western"><surname>Kvedar</surname><given-names>JC</given-names> </name></person-group><article-title>The long but necessary road to responsible use of large language models in healthcare research</article-title><source>NPJ Digit Med</source><year>2024</year><month>07</month><day>4</day><volume>7</volume><issue>1</issue><fpage>177</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01180-y</pub-id><pub-id pub-id-type="medline">38965411</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary methodological tables, including patient inclusion and exclusion criteria, osteonecrosis of the femoral head grading criteria for the Ficat, Association Research Circulation Osseous, and Steinberg systems, prompt-engineering details, and a comparison of multimodal large language model performance with previously published deep learning models for osteonecrosis of the femoral head.</p><media xlink:href="jmir_v28i1e92919_app1.docx" xlink:title="DOCX File, 46 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Supplementary performance and statistical analysis tables, including diagnostic performance, early or late staging performance, intraclass correlation coefficient&#x2013;based reliability results, and statistical comparisons across imaging inputs, prompt configurations, multimodal large language models, and clinical readers.</p><media xlink:href="jmir_v28i1e92919_app2.xlsx" xlink:title="XLSX File, 25 KB"/></supplementary-material></app-group></back></article>