<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e84454</article-id><article-id pub-id-type="doi">10.2196/84454</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Machine Learning to Identify Point-of-Care Ultrasound and Evaluate Standardized Documentation: Retrospective Operational Cohort Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Nguyen</surname><given-names>Kevin</given-names></name><degrees>MS, MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wu</surname><given-names>Zewen</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tsai</surname><given-names>Chu-An</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Vandervest</surname><given-names>John</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lammers</surname><given-names>D&#x2019;Anna</given-names></name><degrees>CPC, BS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cassidy</surname><given-names>Ruth</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Murphy</surname><given-names>Zachary</given-names></name><degrees>MD, MSE</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pandian</surname><given-names>Balaji</given-names></name><degrees>MD, MBA</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hammoud</surname><given-names>Maya M</given-names></name><degrees>MD, MBA</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Collin</surname><given-names>Jennifer</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Smith</surname><given-names>Roger</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Maben-Feaster</surname><given-names>Rosalyn</given-names></name><degrees>MD, MPH</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kaufman Eddy</surname><given-names>Amy</given-names></name><degrees>MPH</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Burns</surname><given-names>Michael L</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Anesthesiology, University of Michigan</institution><addr-line>1500 E. Medical Center Drive</addr-line><addr-line>Ann Arbor</addr-line><addr-line>MI</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Obstetrics and Gynecology, University of Michigan</institution><addr-line>Ann Arbor</addr-line><addr-line>MI</addr-line><country>United States</country></aff><aff id="aff3"><institution>Department of Anesthesiology, Weill Cornell Medicine</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Shivanna</surname><given-names>Abhishek</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Bai</surname><given-names>Chen</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Michael L Burns, MD, PhD, Department of Anesthesiology, University of Michigan, 1500 E. Medical Center Drive, Ann Arbor, MI, 48109, United States, 1 734-936-4280; <email>mlburns@med.umich.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>7</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e84454</elocation-id><history><date date-type="received"><day>19</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>06</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>07</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Kevin Nguyen, Zewen Wu, Chu-An Tsai, John Vandervest, D&#x2019;Anna Lammers, Ruth Cassidy, Zachary Murphy, Balaji Pandian, Maya M Hammoud, Jennifer Collin, Roger Smith, Rosalyn Maben-Feaster, Amy Kaufman Eddy, Michael L Burns. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 7.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e84454"/><abstract><sec><title>Background</title><p>Point-of-care ultrasound (POCUS) is integral to obstetrics and gynecology (OBGYN), offering bedside diagnostic and therapeutic advantages. Despite its widespread adoption, accurate documentation and billing remain challenging due to inconsistent workflows, variable free-text note quality, and inefficiencies within electronic health record (EHR) systems. These barriers often result in missed procedural charges and hinder operational, educational, and reimbursement efforts.</p></sec><sec><title>Objective</title><p>This study leveraged machine learning (ML) to automatically identify POCUS procedures within clinical notes and assessed the effect of implementing standardized procedure documentation (ProcDoc) templates on billing capture accuracy and efficiency.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a multipart retrospective cohort study at a large academic medical center using EHRs from January 2018 to August 2024 across 11 OBGYN clinic sites. ML models (LightGBM [light gradient boosting machine] and BioClinBERT [biomedical and clinical bidirectional encoder representations from transformers]) were trained on clinical encounter notes to classify POCUS procedures and validated against Current Procedural Terminology (CPT) manual code assignments. In February 2023, a standardized ProcDoc smart form was introduced to streamline POCUS documentation and automatically trigger CPT billing codes. Preintervention and postintervention periods were compared using ML metrics and manual billing audits. Outcomes included model accuracy, recall, precision, adoption rates, improvement in billing recapture, and usage of ProcDoc templates.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 559,029 encounters from 109,776 unique patients were analyzed. The BioClinBERT model (accuracy 0.97; <italic>F</italic><sub>1</sub>-score 0.55-0.63) demonstrated a robust ability to identify documented and missed procedures in free-text clinical notes. ProcDoc adoption reached 75.1% within 12 months, supported by comprehensive staff education. Billing recapture&#x2014;the proportion of charges missed by providers but later identified&#x2014;dropped from 10.0% preintervention to 2.4% postintervention, primarily arising from the shift toward auto-capturing documentation (odds ratio 0.22, 95% CI 0.17&#x2010;0.30; <italic>P</italic>&#x003C;.001), with overall POCUS billing slightly increased (+0.6%). Most postintervention CPT codes (1812/2404, 75.4%) originated from ProcDoc templates, confirming improved workflow efficiency and reduced manual audit burden. Model analysis and billing metrics demonstrated that improvements were associated with workflow changes and not an increase in procedure frequency.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>ML modeling proved effective for extracting POCUS procedures from clinical documentation and serving as an evaluation tool for workflow interventions. Standardized documentation with ProcDoc significantly enhanced charge capture accuracy and reduced dependence on manual chart reviews and billing reconciliation. This approach highlights the use of ML as a retrospective auditing and evaluation tool for assessing clinical workflow interventions. Broader application of similar strategies could address documentation inefficiencies and promote sustainability across health care settings.</p></sec></abstract><kwd-group><kwd>point-of-care ultrasound</kwd><kwd>electronic health records</kwd><kwd>medical informatics</kwd><kwd>obstetric labor</kwd><kwd>workflow</kwd><kwd>revenue cycle</kwd><kwd>machine learning</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Point-of-care ultrasound (POCUS) procedures are limited bedside studies that aid in diagnostic assessments and guide therapeutic interventions. These tests are increasingly used in medicine, integrated into training, and widely adopted in obstetrics and gynecology (OBGYN). They differ from traditional ultrasound by addressing a specific clinical question and providing multiple rapid and reliable benefits. POCUS is cost-effective, displays images in real time, and is operated by clinical providers who can correlate the findings with symptoms and provide immediate targeted treatments [<xref ref-type="bibr" rid="ref1">1</xref>]. Proper capture of these procedures requires classification into Current Procedural Terminology (CPT) codes, which are used in research, training, quality improvement, and reimbursement efforts [<xref ref-type="bibr" rid="ref2">2</xref>]. POCUS procedures are susceptible to electronic health record (EHR) capture inefficiencies due to operational variability in training, equipment, and workflows. Appropriate documentation and subsequent charge capture require provider mindfulness and manual review and capture of clinical documentation by billing teams. This is especially true in settings with frequent and widespread use, such as OBGYN visits, leading to inconsistent documentation and uncaptured performed procedures [<xref ref-type="bibr" rid="ref3">3</xref>]. AI applications, such as machine learning (ML), offer an opportunity to identify procedures from clinical documentation. Additionally, simplifying the documentation and billing workflows for these bedside procedures offers the potential for improved capture.</p><p>In OBGYN, proper documentation and reporting of POCUS procedures are critical to reimbursements, research, diagnostic integrity, operations, and legal record-keeping [<xref ref-type="bibr" rid="ref2">2</xref>]. Common reasons leading to inaccuracies include a lack of formal provider education, inadequate clinical documentation, and the absence of quality assurance feedback systems aimed at correcting billing errors [<xref ref-type="bibr" rid="ref3">3</xref>]. In the emergency department, another setting in which POCUS is frequently used, quality improvement interventions have been associated with increased capture rates, reducing unbilled and nonarchived POCUS examinations [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. One study found that only a small minority of emergency medicine practitioners received reimbursement for POCUS from Medicare beneficiaries and hypothesized that most POCUS examinations performed were not billed [<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>POCUS use is documented in the EHR through various methods, including standardized procedure notes, templates, or, more commonly, through free-text documentation in patient clinic visit notes, as is done at our institution. Free-text clinical documentation is variable in quality and contains documentation error rates as high as 10%, increasing the complexity of identifying the use of POCUS from this type of documentation [<xref ref-type="bibr" rid="ref9">9</xref>]. Manual chart review of clinic notes is an arduous task, given the vast amounts of EHR documents generated daily. Advances in natural language processing, a subset of AI, allow automated insights into EHR data, including note classification, entity recognition, and text summarization [<xref ref-type="bibr" rid="ref10">10</xref>]. Identifying examinations from EHR notes falls into an ML task known as classification [<xref ref-type="bibr" rid="ref11">11</xref>], which has previously been successful in health care applications, such as identifying aortic stenosis [<xref ref-type="bibr" rid="ref12">12</xref>], type 2 diabetes [<xref ref-type="bibr" rid="ref13">13</xref>], billing assignments [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>], and predicting miscarriages and stillbirths [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Based on the transformer architecture that underlies large language models, BioBERT (biomedical bidirectional encoder representations from transformers) is a pretrained biomedical language representation model specifically adapted for biomedical text. BioBERT is an extension of the general-purpose language model BERT (bidirectional encoder representations from transformers) and has shown superior performance on biomedical language tasks [<xref ref-type="bibr" rid="ref18">18</xref>]. For certain tasks, studies have shown that a fine-tuned BioBERT model outperforms foundational large language models and logistic regression methods [<xref ref-type="bibr" rid="ref19">19</xref>]. In this multipart retrospective study, we hypothesized that (1) ML modeling could predict OBGYN POCUS procedures from clinical notes, and (2) this modeling could be used to evaluate workflow changes for standardized procedural documentation.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Setting</title><p>This is a multipart retrospective cohort study to (1) develop an ML model for identifying POCUS procedures from clinical notes and (2) use the ML model to evaluate the effectiveness of introducing standardized procedure documentation (ProcDoc) for POCUS procedures. This study used records from 11 unique OBGYN clinic sites within the University of Michigan-Health system, all using the identical EHR system. Four distinct datasets were used in this study: (1) The &#x201C;Development (Dev)&#x201D; dataset was created from all OBGYN visits from January 1, 2018, to December 31, 2020. This dataset alone was used to create the ML model in this study. (2) The first inference set, &#x201C;preintervention,&#x201D; included visits from February 1, 2022, to December 31, 2022. The intervention, described in detail below, occurred in February 2023. (3) The &#x201C;post(1)-intervention&#x201D; dataset included visits from February 1, 2023, to December 31, 2023. (4) The final dataset, &#x201C;post(2)-intervention,&#x201D; included visits from January 1, 2024, to August 31, 2024. The preintervention and post(1)-intervention datasets were aligned with each other and the February intervention, resulting in February to December comparisons and a 1-month gap (January 2023) in the data timeline. There was an approximately 13-month gap between January 2021 and January 2022, during which the OBGYN teams determined next steps to improve POCUS billing capture. Clinic sites slightly differed between the Dev dataset and all other datasets. The Dev dataset consisted of 10 unique clinical sites (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), whereas the preintervention, post(1)-intervention, and post(2)-intervention datasets contained data from 9 clinic sites, 8 of which overlapped with the Dev dataset. One clinic was added, and 2 clinics were removed from the ProcDoc intervention as they had significantly different workflows and EHR capture methods that were incompatible with the ProcDoc intervention. Prior to dataset creation, 2 reviewers (DL and MB) manually identified 4 important clinical free-text note types (model features) used to document POCUS procedures: progress, addendum, procedure, and admission history and physical notes. Datasets were created using all data from these predefined note types during each dataset period, with no additional inclusion or exclusion criteria applied.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This study was approved by the University of Michigan Institutional Review Board with a waiver of informed consent (HUM00203986), and the authors followed the STROBE (Strengthening the Reporting of Observational Studies in Epidemiology) guidelines [<xref ref-type="bibr" rid="ref20">20</xref>] (<xref ref-type="supplementary-material" rid="app2">Checklist 1</xref>). Data were deidentified; however, cases selected for manual review were reidentified solely for this purpose.</p></sec><sec id="s2-3"><title>Data Extraction and Text Preprocessing</title><p>To prepare data for ML modeling, EHR data were collected for each clinic visit encounter and preprocessed as follows. For each encounter, all notes of these types were concatenated and preprocessed by lowercasing, whitespace removal, text tokenization (using the NLTK library [<xref ref-type="bibr" rid="ref21">21</xref>]), text stemming (through the Porter stemmer), and stripping of common words. These steps were performed only for the lightweight gradient-boosting machine (LightGBM) model. BioClinBERT (biomedical and clinical bidirectional encoder representations from transformers) model did not undergo any of these steps. Instead, since BioClinBERT has a sequence length limit of 512 tokens, keyword-based extractive summarization was applied to each document. This step retained only sentences containing at least 1 of 100 specified keywords identified using LightGBM feature importance, which ranks the most influential keywords driving the classification predictions of the LightGBM model (Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) [<xref ref-type="bibr" rid="ref22">22</xref>]. Keywords were initially statistically selected and pruned through manual review (DL and MB) of the sentences captured based on their clinical utility, keeping the top 100 keywords and omitting terms that captured only sentences containing nonclinical artifacts. This was done on the Dev dataset alone and frozen for use across all datasets. After preprocessing, documents that exceeded the sequence length limit were truncated to the first 512 tokens. The stopwords Python library &#x201C;wordcloud&#x201D; was used during preprocessing to determine the top 100 words, but no stopwords were used in the final model preprocessing, thus maintaining both negation cues and clinical context.</p></sec><sec id="s2-4"><title>Model Development and Model Evaluation</title><p>Before the intervention of standardized procedural documentation, POCUS procedures were manually captured from free-text clinic visit notes. To understand this manual capture process, 3 algorithms (2 types of ML models and 1 single-word search method) were developed to classify POCUS procedures from clinical documentation. The word search method simply searched for the term &#x201C;ultrasound&#x201D; in the processed encounter text and was used as a baseline comparison to a simplistic method.</p><p>First, a gradient boosting tree&#x2013;based ML model, LightGBM, was used. Second, we used keyword insights from LightGBM to develop a BioClinBERT model, which builds on BioBERT and has been pretrained using all clinical notes from MIMIC-III (Medical Information Mart for Intensive Care III). Both models used a stratified split based on calendar years, using 80% of the data for training and reserving 20% for hold-out testing. Five-fold cross-validation of the 80% training dataset was used to tune model hyperparameters. During tuning, model weights were initialized using BioBERT-Base (version 1.0), along with PubMed 200K and PMC 270K, and the official tokenizer was adapted specifically for clinical text. Training splits for model development were grouped using encounter ID, independent of patient ID. The maximum input sequence length (including padding) was set to 512 tokens, and optimization was performed using the AdamW optimizer. The selected hyperparameters included a batch size of 16 and a learning rate of 2&#x00D7;10<sup>5</sup>. Final models were retrained using the full development dataset with the selected hyperparameter configuration. The classification threshold was set at the default value of 0.5.</p><p>BioClinBERT was subsequently evaluated on the preintervention and post(1)-intervention datasets. Standard metrics were assessed, including accuracy, recall, precision, and <italic>F</italic><sub>1</sub>-score. The gold-standard reference label for whether a POCUS procedure was performed during an encounter was determined by the presence of one or more of 12 POCUS-specific CPT codes: 58340, 76801, 76802, 76815&#x2010;76819, 76830, 76831, 76856, and 76857 (Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Manual validation was conducted by a certified clinical documentation specialist (DL) on a random sample of encounters in which the model predicted a procedure, but no corresponding POCUS CPT codes were billed (&#x201C;false positives&#x201D;) and on another random sample of encounters in which the model failed to identify a billed POCUS procedure (&#x201C;false negatives&#x201D;).</p></sec><sec id="s2-5"><title>ProcDoc Intervention</title><p>From the preintervention evaluation, the OBGYN group identified 9 clinics for workflow interventions, 8 of which were included in the Dev dataset (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The OBGYN team designed and implemented standardized POCUS procedure documentation (ProcDoc) templates (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The intervention occurred in February 2023 using Epic Smart Forms (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), which provide customizable fields and selections for capturing indications, impressions, and other relevant details. The ProcDoc was designed to be used with OBGYN POCUS procedures and, once completed, was organized under the Procedures tab in the EHR. Submission of the ProcDoc automatically triggered the associated CPT billing code. Staff education coincided with the rollout and included formal presentations, video demonstrations, and ad hoc peer-to-peer teaching. Education was disseminated during staff meetings to faculty, residents, advanced practice providers, and certified nurse midwives, and was included in onboarding for new hires and new resident cohorts. Metrics on individual provider usage were collected, and reminders were sent to providers who displayed low rates of ProcDocs usage. The ProcDoc template was not mandatory to close an encounter. Reports on provider usage of the ProcDoc were compiled periodically. If a provider was identified as not using the ProcDoc, they were sent a tipsheet and reminder notifying them that ProcDoc was the preferred method for documentation and charge capture of clinic-performed POCUS examinations.</p></sec><sec id="s2-6"><title>ProcDoc Intervention Evaluation and Statistical Analyses</title><p>There were 2 main evaluations: ML-based and billing team&#x2013;based evaluations. ML-based evaluations used the Dev, preintervention, and post(1)-intervention datasets, whereas billing team&#x2013;based evaluations used the preintervention, post(1)-intervention, and post(2)-intervention datasets. For ML-based evaluations, the BioClinBERT ML model was used to label encounters. BioClinBERT was not applied to the post(2)-intervention dataset, as this dataset existed in the OBGYN billing evaluations and was inconsistent with the modeling time periods for this study. Following initial training on the Dev dataset, this model remained fixed throughout the evaluations of the preintervention and post(1)-intervention datasets.</p><p>For billing team&#x2013;based evaluations, the OBGYN billing team tracked the following 3 key metrics: (1) total outpatient ultrasound billing&#x2014;charges for the 2 most common OBGYN POCUS CPT codes (76815 and 76857), (2) ultrasound billing recaptures&#x2014;charges that were originally missed by providers but later recaptured by the billing reconciliation and audit teams, and (3) ProcDoc adoption rates&#x2014;the proportion of ultrasound charges derived from ProcDocs out of all charges. Review recapture was used as an objective outcome to estimate efficiency improvement from the ProcDoc intervention conducted by the OBGYN billing team and was determined separately from ML results and evaluations. The methodology used by the billing team involved a manual review consisting of reading each note to find missing POCUS charges. They reviewed every note affiliated with each encounter to determine appropriate charges to capture. The intensity, staffing, specific ultrasound procedural focus, reconciliation processes, and audit workflows of the billing team&#x2019;s review process remained stable throughout the study.</p><p>Billing team metrics were assessed on the preintervention, post(1)-intervention, and post(2)-intervention data. To compare distributional summaries of covariates between datasets, we calculated pairwise standardized differences. An absolute standardized difference threshold exceeding 0.2 was used to indicate imbalance. We calculated the rates and odds of recapture for the preintervention, post(1)-intervention, and post(2)-intervention time periods. We then calculated odds ratios, along with corresponding 95% CIs and <italic>P</italic> values, for both postintervention time periods, using the preintervention period as the reference. For ease of interpretability, we also reported the inverse of the odds ratio estimates. Rates and CIs were adjusted to account for repeated measures by patient through a mixed-effects logistic model with random intercepts corresponding to the unique patient identifier and a compound symmetry covariance structure.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Dataset Characteristics</title><p>A total of 559,029 clinical encounters from 109,776 unique patients were included across the 4 datasets: Dev, preintervention, post(1)-intervention, and post(2)-intervention (<xref ref-type="fig" rid="figure1">Figure 1</xref>; <xref ref-type="table" rid="table1">Table 1</xref>). The average patient age ranged from 36.0 (SD 12.2) to 38.5 (SD 14.1) years, with no significant differences between datasets. Between 10.6% (33,643/316,390) and 11.5% (10,978/95,696) encounters were identified as Black patients, and between 4.3% (13,714/316,390) and 5.4% (4980/91,627) were identified as Hispanic. The POCUS rate identified by CPT codes was 9.9% (31,336/316,390 encounters) in the Dev dataset and ranged from 3.6% (3414/95,696) to 4.3% (2402/55,316) in the other datasets, a difference attributed to clinic inclusion and exclusion after model development (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). ProcDoc implementation focused on primary OBGYN clinics, which omitted 2 clinics included in the Dev dataset and added a clinic that was not included in the Dev dataset. All absolute standardized differences between datasets were &#x2264;0.2.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study flowchart for AI model development, inference, and obstetrics and gynecology (OBGYN) billing team evaluations. A single AI model was developed and used to evaluate the preintervention and post(1)-intervention data. The intervention included standardized documentation (ProcDoc) implementation and clinical workflow education. OBGYN billing team evaluations spanned preintervention, post(1)-intervention, and post(2)-intervention datasets.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84454_fig01.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Dataset characteristics<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristics</td><td align="left" valign="bottom">Development,<break/>Jan 2018 to<break/>Dec 2020</td><td align="left" valign="bottom">Preintervention,<break/>Feb 2022 to<break/>Dec 2022</td><td align="left" valign="bottom">Post(1)-intervention,<break/>Feb 2023 to<break/>Dec 2023</td><td align="left" valign="bottom">Post(2)-intervention,<break/>Jan 2024 to<break/>Aug 2024</td><td align="left" valign="bottom" colspan="4">Standardized difference</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom">Pre to<break/>Dev</td><td align="left" valign="bottom">Post(1) to<break/>Dev</td><td align="left" valign="bottom">Post(1) to<break/>Pre</td><td align="left" valign="bottom">Post(2) to Post(1)</td></tr></thead><tbody><tr><td align="left" valign="bottom">Patients, n<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="bottom">68,809</td><td align="left" valign="bottom">40,132</td><td align="left" valign="bottom">40,429</td><td align="left" valign="bottom">30,963</td><td align="left" valign="middle">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="bottom">&#x2014;</td><td align="left" valign="bottom">&#x2014;</td><td align="left" valign="bottom">&#x2014;</td></tr><tr><td align="left" valign="top">Encounters, n</td><td align="left" valign="top">316,390</td><td align="left" valign="top">95,696</td><td align="left" valign="top">91,627</td><td align="left" valign="top">55,316</td><td align="left" valign="middle">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">POCUS rate, encounter n (%)<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">31,336 (9.9)</td><td align="left" valign="top">3414 (3.6)</td><td align="left" valign="top">3358 (3.7)</td><td align="left" valign="top">2402 (4.3)</td><td align="left" valign="middle">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Note types per encounter, mean (SD)</td><td align="left" valign="top">1.2 (0.4)</td><td align="left" valign="top">1.1 (0.3)</td><td align="left" valign="top">1.1 (0.4)</td><td align="left" valign="top">1.2 (0.4)</td><td align="left" valign="middle">&#x2212;0.17</td><td align="left" valign="top">&#x2212;0.08</td><td align="left" valign="top">0.09</td><td align="left" valign="top">0.2</td></tr><tr><td align="left" valign="top">Age (y) mean (SD); missing n (%)</td><td align="left" valign="top">36.0 (12.2); 18 (0)</td><td align="left" valign="top">36.7 (13.2); 27 (0)</td><td align="left" valign="top">37.5 (13.5); 33 (0)</td><td align="left" valign="top">38.5 (14.1); 0 (0)</td><td align="left" valign="middle">0.06</td><td align="left" valign="top">0.12</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.08</td></tr><tr><td align="left" valign="top">Weight (lbs), mean (SD); missing n (%)<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top">173.2 (45.5); 64,913 (20.5)</td><td align="left" valign="top">175.0 (45.4); 19,275 (20.1)</td><td align="left" valign="top">176.5 (46.2); 17,824 (19.5)</td><td align="left" valign="top">176.1 (46.1); 12,278 (22.2)</td><td align="left" valign="middle">0.04</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.03</td><td align="left" valign="top">&#x2212;0.01</td></tr><tr><td align="left" valign="top">Height (inches), mean (SD); missing n (%)<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top">64.5 (3.0); 176,892 (55.9)</td><td align="left" valign="top">64.5 (2.9); 60,211 (62.9)</td><td align="left" valign="top">64.5 (3.0); 60,723 (66.3)</td><td align="left" valign="top">64.4 (3.1); 38,409 (69.4)</td><td align="left" valign="middle">0.01</td><td align="left" valign="top">&#x2212;0.01</td><td align="left" valign="top">&#x2212;0.01</td><td align="left" valign="top">&#x2212;0.02</td></tr><tr><td align="left" valign="top">BMI, median (IQR); missing n (%)<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top">27.8 (24.0-32.9);<break/>64,669 (20.4)</td><td align="left" valign="top">28.2 (24.2-33.5); 20,542 (21.5)</td><td align="left" valign="top">28.4 (24.4-33.7); 19,763 (21.6)</td><td align="left" valign="top">28.4 (24.3-33.7); 13,868 (25.1)</td><td align="left" valign="middle">0.05</td><td align="left" valign="top">0.08</td><td align="left" valign="top">0.03</td><td align="left" valign="top">&#x2212;0.004</td></tr><tr><td align="left" valign="top">Race, encounter n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="middle">0.05</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.04</td><td align="left" valign="top">0.02</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>White</td><td align="left" valign="top">240,663 (76.1)</td><td align="left" valign="top">72,335 (75.6)</td><td align="left" valign="top">68,919 (75.2)</td><td align="left" valign="top">41,647 (75.3)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Black</td><td align="left" valign="top">33,643 (10.6)</td><td align="left" valign="top">10,978 (11.5)</td><td align="left" valign="top">10,196 (11.1)</td><td align="left" valign="top">6012 (10.9)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Asian</td><td align="left" valign="top">23,880 (7.5)</td><td align="left" valign="top">6295 (6.6)</td><td align="left" valign="top">5885 (6.4)</td><td align="left" valign="top">3615 (6.5)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>American Indian</td><td align="left" valign="top">14,285 (4.5)</td><td align="left" valign="top">4749 (5.0)</td><td align="left" valign="top">5114 (5.6)</td><td align="left" valign="top">3031 (5.5)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Unknown, missing, or refused</td><td align="left" valign="top">3919 (1.2)</td><td align="left" valign="top">1339 (1.4)</td><td align="left" valign="top">1513 (1.7)</td><td align="left" valign="top">1011 (1.8)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Ethnicity, encounter n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="middle">0.03</td><td align="left" valign="top">0.05</td><td align="left" valign="top">0.02</td><td align="left" valign="top">0.03</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hispanic</td><td align="left" valign="top">13,714 (4.3)</td><td align="left" valign="top">4761 (5.0)</td><td align="left" valign="top">4980 (5.4)</td><td align="left" valign="top">2879 (5.2)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Non-Hispanic</td><td align="left" valign="top">295,702 (93.5)</td><td align="left" valign="top">88,779 (92.8)</td><td align="left" valign="top">84,455 (92.2)</td><td align="left" valign="top">50,889 (92.0)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Unknown, missing, refused</td><td align="left" valign="top">6974 (2.2)</td><td align="left" valign="top">2156 (2.3)</td><td align="left" valign="top">2192 (2.4)</td><td align="left" valign="top">1548 (2.8)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Payer, encounter n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="middle">0.12</td><td align="left" valign="top">0.15</td><td align="left" valign="top">0.03</td><td align="left" valign="top">0.04</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BCBS<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="left" valign="top">222,034 (70.2)</td><td align="left" valign="top">64,264 (67.2)</td><td align="left" valign="top">59,909 (65.4)</td><td align="left" valign="top">36,076 (65.2)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Commercial</td><td align="left" valign="top">70,056 (22.1)</td><td align="left" valign="top">26,005 (27.2)</td><td align="left" valign="top">25,962 (28.3)</td><td align="left" valign="top">15,868 (28.7)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Military</td><td align="left" valign="top">1439 (0.5)</td><td align="left" valign="top">478 (0.5)</td><td align="left" valign="top">406 (0.4)</td><td align="left" valign="top">157 (0.3)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medicaid</td><td align="left" valign="top">4827 (1.5)</td><td align="left" valign="top">893 (0.9)</td><td align="left" valign="top">914 (1.0)</td><td align="left" valign="top">639 (1.2)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medicare</td><td align="left" valign="top">12,023 (3.8)</td><td align="left" valign="top">3446 (3.6)</td><td align="left" valign="top">3436 (3.7)</td><td align="left" valign="top">1840 (3.3)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Worker&#x2019;s comp</td><td align="left" valign="top">5 (0.0)</td><td align="left" valign="top">0 (0.0)</td><td align="left" valign="top">3 (0.0)</td><td align="left" valign="top">3 (0.0)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Unknown or Missing</td><td align="left" valign="top">6006 (1.9%)</td><td align="left" valign="top">610 (0.6%)</td><td align="left" valign="top">997 (1.1%)</td><td align="left" valign="top">733 (1.3%)</td><td align="left" valign="middle"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Dataset demographics: Dev (development), Pre (preintervention), and Post(1) and Post(2) (2 postinterventions) datasets. Race categories: White (White or Caucasian only), Black (Black or African American), Asian (Asian or Pacific Islander, not Black, not Native), and American Indian (American Indian and Alaska Native or Middle Eastern/North African or Other). The Dev point-of-care ultrasound (POCUS) rate examined the rate of ultrasounds performed in the clinics within the Dev dataset, whereas the pre-post POCUS rate analyzed clinics from the preintervention and postintervention datasets and extended to all other datasets. Demographic variables were neither incorporated into the machine learning pipeline nor used in billing metrics.</p></fn><fn id="table1fn2"><p><sup>b</sup>A total of 109,776 unique patients are represented across all datasets.</p></fn><fn id="table1fn3"><p><sup>c</sup>Not applicable.</p></fn><fn id="table1fn4"><p><sup>d</sup>POCUS rate variation is based on variations in clinic inclusion (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></fn><fn id="table1fn5"><p><sup>e</sup>BMI was recorded independent of height and weight in these patient populations.</p></fn><fn id="table1fn6"><p><sup>f</sup>BCBS: Blue Cross Blue Shield.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Model Development and Preintervention Evaluation</title><p>The same ML model was applied without retraining at any stage of the project to retain focus and consistency throughout preintervention and postintervention comparisons. Using the Dev dataset, 3 approaches were evaluated to predict OBGYN POCUS procedures from free-text in clinical notes: word search, LightGBM, and BioClinBERT. The intention of using a word search was to compare a simple, unoptimized search strategy. Word search yielded accuracy, recall, precision, and <italic>F</italic><sub>1</sub>-score of 0.86, 0.59, 0.36, and 0.44, respectively, whereas LightGBM (0.95, 0.90, 0.65, and 0.76, respectively) and BioClinBERT (0.97, 0.79, 0.86, and 0.82, respectively) performed better. Results analyzed across distinct OBGYN locations showed an average site accuracy of 0.93 (SD 0.11), with a maximum of 0.98 and a minimum of 0.67. Only 2 of 10 sites showed a model accuracy of less than 95%, each containing the lowest number of patient encounters; A total of 9.9% (n=31,336) of the Dev dataset encounters contained one or more of the associated CPT codes (<xref ref-type="table" rid="table1">Table 1</xref>; Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Model results varied by clinic (Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). BioClinBERT ML modeling labeled an additional 1.1% of encounters as positive, representing an 11.6% increase compared to CPT codes. The gold-standard label for model training used existing billing codes, which contained missed billing encounters and thus are variable and potentially unreliable. While it is infeasible to manually review notes from each encounter, validation was conducted (by reviewer DL) on simple random samples from the Dev dataset for false positives and false negatives. False positive samples consisted of 101 encounters, where the model predicted a procedure, but no associated POCUS CPT codes were billed. Of these 101 encounters, 85 (84.2%) encounters (adjusted 84.0% with 95% CI 76.8%-91.2%) were confirmed as correct upon manual review, indicating that BioClinBERT successfully identified missed POCUS procedures not captured in billing data. A random sample of 114 false negative cases was reviewed, and 63 (55.3%; adjusted 54.9%, 95% CI 45.7%-64.1%) were found to be truly negative for a POCUS procedure.</p></sec><sec id="s3-3"><title>ProcDoc Intervention and Evaluation</title><p>Adoption of ProcDoc rose rapidly after implementation, reaching an adoption rate of 64.2% in August 2023, 75.1% in February 2024, and 77.9% in August 2024 (<xref ref-type="fig" rid="figure2">Figure 2</xref>). Relative to preintervention, post(1) data were associated with improved true positive labeling, as evidenced by increased precision (0.562 vs 0.487), recall (0.711 vs 0.635), and positive likelihood ratio (36.2 vs 28.2), indicating improved true positive labeling after the intervention (<xref ref-type="table" rid="table2">Table 2</xref>). There was sequential improvement over time following the intervention (Figures S1-S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). False positives (model prediction positive but no CPT codes billed) were considered a flag for possible missed billing opportunities, identifying cases that could potentially be examined further to identify missed billing charges. The false discovery rate was calculated as the percentage of cases where no CPT codes were billed out of those predicted positive in the model. However, the nature of false positives could shift over time or across different clinics, making this an imperfect measure. Post(1) encounters displayed a decreased false discovery rate (0.438 vs 0.513). Trending by month again displayed appreciable and sustained improvement postintervention (Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The OBGYN billing team evaluated the 2 most common POCUS CPT codes: 76815 and 76857 (<xref ref-type="table" rid="table3">Table 3</xref>). These 2 codes represented more than 95% of all encounters from the training dataset. The total ultrasounds billed slightly increased from 3176 to 3195 over the same time period, preintervention to post(1)-intervention. CPT code assignment occurred during 2 distinct phases: (1) &#x201C;Original,&#x201D; where codes were assigned directly by providers based on initial documentation, and (2) &#x201C;Review,&#x201D; where codes were added later through manual review by billing reconciliation and audit teams. From preintervention to post(1)-intervention to post(2)-intervention, the percentage of review recapture decreased from 10.0% (317/3176) to 2.4% (58/2404), while capture increased, primarily due to the shift toward auto-capturing documentation, with 75.4% (1812/2404) of CPT code captures originating from ProcDocs (<xref ref-type="table" rid="table3">Table 3</xref>). The odds of review recapture for the post(2)-intervention group were 0.22 (95% CI 0.17-0.30; <italic>P</italic>&#x003C;.001) compared to the preintervention group.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Procedure document (ProcDoc) adoption rate following the February 2023 implementation workflow intervention.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84454_fig02.png"/></fig><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Machine learning (ML) model statistical output measurements comparing preintervention and postintervention data by encounter<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dataset</td><td align="left" valign="bottom" colspan="2">ML model predicted<break/>positive</td><td align="left" valign="bottom" colspan="2">ML model predicted negative</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Precision rate (positive predictive value) = TP<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> / predicted positive</td><td align="left" valign="bottom">Recall<break/>(sensitivity) = TP / positive</td><td align="left" valign="bottom">Specificity (true negative rate) =<break/>TN<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup> / negative</td><td align="left" valign="bottom">Positive likelihood ratio = recall / FP<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup> rate</td><td align="left" valign="bottom">False<break/>discovery rate = FP / predicted positive</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Billing Data POCUS CPT<break/>Yes<break/>(TP<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>)</td><td align="left" valign="bottom">Billing Data POCUS CPT<break/>No<break/>(FP<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>)</td><td align="left" valign="bottom">Billing Data POCUS CPT<break/>Yes<break/>(FN<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup>)</td><td align="left" valign="bottom">Billing Data POCUS CPT<break/>No<break/>(TN<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup>)</td><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/></tr></thead><tbody><tr><td align="left" valign="top">Preintervention</td><td align="left" valign="top">1979</td><td align="left" valign="top">2084</td><td align="left" valign="top">1137</td><td align="left" valign="top">90,424</td><td align="left" valign="top">0.966</td><td align="left" valign="top">0.551</td><td align="left" valign="top">0.487</td><td align="left" valign="top">0.635</td><td align="left" valign="top">0.977</td><td align="left" valign="top">28.2</td><td align="left" valign="top">0.513</td></tr><tr><td align="left" valign="top">Post(1)-intervention</td><td align="left" valign="top">2199</td><td align="left" valign="top">1717</td><td align="left" valign="top">894</td><td align="left" valign="top">85,776</td><td align="left" valign="top">0.971</td><td align="left" valign="top">0.627</td><td align="left" valign="top">0.562</td><td align="left" valign="top">0.711</td><td align="left" valign="top">0.980</td><td align="left" valign="top">36.2</td><td align="left" valign="top">0.438</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Predicted positive is TP+FP, negative is FP+TN, positive is FN+TP. Positive was determined by the presence of one or more of 12 POCUS-specific current procedural terminology (CPT) codes: 58340, 76801, 76802, 76815-76819, 76830, 76831, 76856, and 76857 (Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></fn><fn id="table2fn2"><p><sup>b</sup>TP: true positive.</p></fn><fn id="table2fn3"><p><sup>c</sup>TN: true negative.</p></fn><fn id="table2fn4"><p><sup>d</sup>FP: false positive.</p></fn><fn id="table2fn5"><p><sup>e</sup>FN: false negative.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Billing team evaluations by encounter over 3 windows of time: preintervention, post(1)-intervention, and post(2)-intervention<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">OBGYN<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup> billing team evaluations</td><td align="left" valign="bottom">Preintervention, Jan 2022 to Dec 2022</td><td align="left" valign="bottom">Post(1)-intervention, Jan 2023 to Dec 2023</td><td align="left" valign="bottom">Post(2)-intervention, January 1, 2024, to August 31, 2024</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">CPT<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> code: 76815</td></tr><tr><td align="left" valign="top" colspan="4">Original</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>non-ProcDoc, n</td><td align="left" valign="top">2803</td><td align="left" valign="top">1359</td><td align="left" valign="top">519</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ProcDoc, n</td><td align="left" valign="top">0</td><td align="left" valign="top">1570</td><td align="left" valign="top">1795</td></tr><tr><td align="left" valign="top">Review, n</td><td align="left" valign="top">317</td><td align="left" valign="top">171</td><td align="left" valign="top">57</td></tr><tr><td align="left" valign="top" colspan="4">CPT code: 76857</td></tr><tr><td align="left" valign="top" colspan="4">Original</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>non-ProcDoc, n</td><td align="left" valign="top">56</td><td align="left" valign="top">63</td><td align="left" valign="top">15</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ProcDoc, n</td><td align="left" valign="top">0</td><td align="left" valign="top">31</td><td align="left" valign="top">17</td></tr><tr><td align="left" valign="top">Review, n</td><td align="left" valign="top">0</td><td align="left" valign="top">1</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">Total capture (original+review), n</td><td align="left" valign="top">3176</td><td align="left" valign="top">3195</td><td align="left" valign="top">2404</td></tr><tr><td align="left" valign="top">Review recapture, n (%)</td><td align="left" valign="top">317 (10.0)</td><td align="left" valign="top">172 (5.4)</td><td align="left" valign="top">58 (2.4)</td></tr><tr><td align="left" valign="top">Odds of review recapture</td><td align="left" valign="top">0.111</td><td align="left" valign="top">0.057</td><td align="left" valign="top">0.025</td></tr><tr><td align="left" valign="top">Odds ratio of review recapture (ref: preintervention)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">0.513</td><td align="left" valign="top">0.224</td></tr><tr><td align="left" valign="top">95% CI (ref: preintervention)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.422-0.623</td><td align="left" valign="top">0.168-0.298</td></tr><tr><td align="left" valign="top"><italic>P</italic> value (ref: preintervention)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Inverse odds ratio of review recapture (ref: preintervention)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">1.949</td><td align="left" valign="top">4.464</td></tr><tr><td align="left" valign="top">ProcDoc capture, n (%)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">1601 (50.1)</td><td align="left" valign="top">1812 (75.4)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Manual internal billing team evaluations for capture consisted of original evaluations and reviews. Point-of-care ultrasound (POCUS) use determines notes using the standardized documentation (ProcDoc) in the process. Adjusted estimates were calculated to account for clustering within patients. Using the subsample (8775 encounters corresponding to 7032 patients) used to calculate recapture odds, odds ratios, 95% CI, and <italic>P</italic> values, we fit a mixed-effects logistic model with random intercepts corresponding to unique patient identifiers using a compound symmetry covariance structure. The variance parameter estimate corresponding to the random effect of the mixed-effects logistic model was 0.068, with an SE of 0.02. Review was conducted for the 2 most common obstetrics and gynecology POCUS CPT codes (76815 and 76857; Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></fn><fn id="table3fn2"><p><sup>b</sup>CPT: Current Procedural Terminology.</p></fn><fn id="table3fn3"><p><sup>c</sup>OBGYN: obstetrics and gynecology.</p></fn><fn id="table3fn4"><p><sup>d</sup>Reference, so not applicable.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this study, we first developed an ML model based on the BERT architecture to identify POCUS procedures in OBGYN clinical documentation and demonstrated its effectiveness across multiple clinics within a single, large academic medical center. Second, we used this ML model to help evaluate a workflow intervention incorporating standardized documentation, with results associated with significantly improved capture of POCUS procedures. Improved capture, evidenced by both ML model performance and billing team evaluation metrics, reduced the existing manual efforts required from billing teams dedicated to this area.</p><p>This study demonstrated the use of AI to guide and evaluate clinical workflow improvements. This finite use of AI technology is unique in that health care applications of AI are primarily continuous and require EHR integration, monitoring, and maintenance solutions. This work exemplifies how AI-driven clinical workflow analyses can address systemic inefficiencies without requiring direct AI model integration and the associated costs. The transition to ProcDocs significantly improved the capture rates of POCUS procedures. This intervention not only streamlined documentation but also ensured that relevant CPT codes were automatically mapped and triggered, reducing the administrative burden, decreasing the need for recapture, and minimizing the chances of missed billing opportunities. This reduction in recapture reflects a shift in documentation volume toward auto-capturing templates that reduced the absolute manual audit burden. The minimal increase in POCUS use from preintervention to postintervention suggests that the workflow standardization was a major contributor to improved capture. The rapid adoption of the ProcDoc by clinical providers, supported by comprehensive staff education and continuous monitoring, indicates the practicality of this type of intervention.</p><p>ML modeling was not necessary to identify the challenges with billing capture, but it aided the process. Subsequently, the modeling we developed was useful for evaluating the workflow intervention and served as a backup in the event that the workflow intervention was unsuccessful. Our modeling results were strong relative to previous studies using AI in health care billing prediction. Specifically, AI to identify <italic>ICD</italic> (<italic>International Classification of Diseases</italic>) codes has been associated with low agreement with human coders (~15%) [<xref ref-type="bibr" rid="ref23">23</xref>], while identification of anesthesiology CPT codes yielded accuracies as high as ~88% [<xref ref-type="bibr" rid="ref16">16</xref>]. The BioClinBERT model developed in this study met or surpassed these previously reported performance metrics. These modeling results underscore the ability to classify complex medical narratives accurately, highlighting the potential of AI to enhance clinical documentation and classification. Manual chart reviews, although time-intensive and impractical due to the sheer volume of clinical notes, can be augmented or replaced by scalable and efficient AI models. The use of AI in this study was as a finite evaluation tool, where observed changes in capture were attributed to the ProcDoc intervention rather than AI modeling. This method of AI is low-cost and yielded great results. Alternatively, when using AI tools in continuous operations, costs to develop and maintain ML models can outweigh potential modest increases in captured billing [<xref ref-type="bibr" rid="ref24">24</xref>]. One study estimated US $217,000 to develop and US $6000 to maintain each ML model. At approximately US $100 per POCUS study, this would require the capture of 60 additional POCUS exams for maintenance and 2170 for model development costs. While precise institutional costs were not formally tracked, model development required dedicated data engineering effort and secure computing infrastructure typical of academic ML workflows. While this work focuses on capturing missed revenue opportunities through workflow changes, in general, the use of AI tools could identify baseline over-billing in the human reference standard and thereby help to mitigate institutional audit risk. Specifically, the manual audit indicated a substantial error rate in the billing reference standard; it simultaneously confirmed that the BioClinBERT model failed to identify an ultrasound procedure in 51 of 114 (44.7%) of the audited false negative encounters. These considerations are imperative for return-on-investment estimations when considering AI applications.</p></sec><sec id="s4-2"><title>Comparison to Prior Work</title><p>Finally, this work expands AI applications in health care, expanding beyond obstetrics, yet adding to previous predictive AI research in the area, such as predicting miscarriages and stillbirths [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>], successful vaginal deliveries [<xref ref-type="bibr" rid="ref25">25</xref>], vaginal birth after cesarean delivery [<xref ref-type="bibr" rid="ref26">26</xref>], acidemia at birth [<xref ref-type="bibr" rid="ref27">27</xref>], postpartum depression [<xref ref-type="bibr" rid="ref28">28</xref>], and equitable considerations for the use of AI [<xref ref-type="bibr" rid="ref29">29</xref>]. The results of this study are applicable to health care billing at large, where the quality of documentation and capture is immensely important for organizational sustainability and quality of care [<xref ref-type="bibr" rid="ref30">30</xref>]. Reliable capture and reporting of clinical procedures are crucial for patient care, diagnostic accuracy, operational efficiency, and appropriate reimbursement&#x2014;key factors in both the success and quality of health care services. As interest in AI-driven payer denial processes grows, further research is needed to integrate AI into health care revenue cycle operations, particularly in coding and billing workflows, to improve financial sustainability and streamline reimbursement [<xref ref-type="bibr" rid="ref31">31</xref>].</p></sec><sec id="s4-3"><title>Limitations</title><p>Despite considerable improvements using this strategy, several limitations exist. (1) The study was conducted on a single procedure type, within a single health care system, using a single EHR vendor, which may limit the generalizability of findings to other clinical environments with different documentation practices and procedures. The performance of this ML model may vary across different clinics and may worsen when extrapolated to other clinics or institutions. Specifically, the ML development dataset included clinics with disproportionately high POCUS use, from which the model will likely perform better with documentation patterns specific to these settings. (2) Our ML model relied on the quality and consistency of input data, which may differ across settings due to variations in EHR maintenance, clinician documentation, billing workflows, and EHR vendor services. (3) As data changes over time due to drift and shift, ML models will be time-sensitive without retraining. (4) Our comparison of models was rigorous and iterative but not exhaustive, and there may exist improved ML models for the classification of OBGYN POCUS procedures. (5) The use of keyword extractive summarization risks reducing key components for model inputs, resulting in model underperformance. Additionally, the extractive summarization may have retained nonclinical terms and may be susceptible to shortcut associations. (6) A few clinics changed between the training and evaluation periods, causing a reduction in the POCUS+ rates in the development and subsequent datasets. This reduction could lead to misbalanced datasets that are less representative of the model training dataset, causing temporal biases in the modeling, specifically concerning overfitting training datasets and high-volume clinics. (7) There exist noncontiguous time windows and multicomponent interventions in this study, and other time-varying factors and cointerventions may contribute to the observed reductions in recapture and improved model-CPT agreement. (8) While false positives of the model may be interpreted as potential missed billing opportunities and validated with a small case review of randomly sampled cases, more exact determinations of missed billing rates would require extensive records reviews that were beyond the scope of this project and our institution&#x2019;s current capacity. (9) Training splits for model development were grouped using encounter ID, independent of patient ID. As a patient may have multiple encounters within each dataset, the same patient may be represented multiple times across different datasets, though each encounter was unique. (10) As the keyword vocabulary was fixed using 2018 to 2020 data, the method may be sensitive to temporal drift in clinical language, potentially missing newer terms or language changes introduced after 2020. (11) Billing team evaluations addressed only the 2 most common CPT codes, omitting approximately 5% of codes used for this type of exam. (12) Analysis of administrative improvements is strictly conditional on an encounter eventually being billed, leaving the true baseline of total procedures performed unknown. (13) The comparison against a single-term word search may overstate the apparent performance gap relative to a more comprehensive clinical search string. (14) Finally, restricting model input to 4 note types may cap the model&#x2019;s recall and inflate the apparent false negative count relative to institution-wide billing capture.</p></sec><sec id="s4-4"><title>Future Directions</title><p>Future research should aim to validate these findings across diverse health care settings and specialties to establish broader applicability. Our findings may encourage similar interventions in other medical departments where procedural documentation is critical. We advocate for continued exploration and integration of AI tools in clinical practice to enhance accuracy, efficiency, and sustainability in health care delivery. Specifically, applying ML for real-time detection of missed charge capture opportunities holds promise for further reduction of manual labor and improvement of administrative efficiency.</p></sec><sec id="s4-5"><title>Conclusions</title><p>ML modeling proved effective for extracting POCUS procedures from clinical documentation and serving as an evaluation tool for workflow interventions. Standardized documentation with ProcDoc significantly enhanced accurate charge capture and reduced dependence on manual chart reviews and billing reconciliation. This approach highlights the use of ML as a retrospective auditing and evaluation tool for assessing clinical workflow interventions. Broader application of similar strategies could address documentation inefficiencies and promote sustainability across health care settings.</p></sec></sec></body><back><ack><p>We express our sincere thanks to the Anesthesiology and Obstetrics and Gynecology departments, the Michigan Anesthesiology Informatics and Systems Improvement Exchange, and the Artificial Intelligence Group at the University of Michigan for their technical, statistical, and preparatory support for this project.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec><sec><title>Data Availability</title><p>The datasets generated and/or analyzed during this study are not publicly available, as the datasets involved in this study are defined as limited datasets per United States Federal Regulations and require the execution of a data use agreement for the transfer or use of the data. The investigative team is able to share data securely and transparently upon reasonable request to the corresponding author, conditional on (1) receipt of a detailed written request identifying the requestor, purpose, and proposed use of the shared data; (2) use of a secure enclave for the sharing of personally identifiable information; and (3) the request being permissible within the confines of existing data use agreements at the institutions.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: CAT, DL, AKE, MLB</p><p>Data curation: ZW, CAT, JV, BP</p><p>Formal analysis: ZW, CAT, JV, BP, RC</p><p>Investigation: KN, DL, AKE, MLB</p><p>Resources: CAT</p><p>Software: ZW, CAT</p><p>Supervision: JV, AKE, MLB</p><p>Validation: CAT, RC, MLB</p><p>Visualization: KN, DL, RC, MLB</p><p>Writing &#x2013; original draft: KN, ZW, CAT, JV, DL, RC, ZM, BP, MMH, JC, RS, RM-F, AKE, MLB</p><p>Writing &#x2013; review &#x0026; editing: KN, ZW, CAT, JV, DL, RC, ZM, BP, MMH, JC, RS, RM-F, AKE, MLB</p></fn><fn fn-type="conflict"><p>MB and JV are coinventors on patent number 11,288,445 B2, entitled &#x201C;Automated System and Method for Assigning Billing Codes to Medical Procedures,&#x201D; related to the use of machine learning techniques for medical procedural billing. MB and JV reported holding equity in the company Decimal Code. No other disclosures were reported.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERT</term><def><p>bidirectional encoder representations from transformers</p></def></def-item><def-item><term id="abb2">BioClinBERT</term><def><p>biomedical and clinical bidirectional encoder representations from transformers</p></def></def-item><def-item><term id="abb3">CPT</term><def><p>Current Procedural Terminology</p></def></def-item><def-item><term id="abb4">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb5"><italic>ICD</italic></term><def><p><italic>International Classification of Diseases</italic></p></def></def-item><def-item><term id="abb6">LightGBM</term><def><p>light gradient-boosting machine</p></def></def-item><def-item><term id="abb7">MIMIC-III</term><def><p>Medical Information Mart for Intensive Care III</p></def></def-item><def-item><term id="abb8">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb9">OBGYN</term><def><p>obstetrics and gynecology</p></def></def-item><def-item><term id="abb10">POCUS</term><def><p>point-of-care ultrasound</p></def></def-item><def-item><term id="abb11">ProcDoc</term><def><p>procedure documentation</p></def></def-item><def-item><term id="abb12">STROBE</term><def><p>Strengthening the Reporting of Observational Studies in Epidemiology</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Recker</surname><given-names>F</given-names> </name><name name-style="western"><surname>Weber</surname><given-names>E</given-names> </name><name name-style="western"><surname>Strizek</surname><given-names>B</given-names> </name><name name-style="western"><surname>Gembruch</surname><given-names>U</given-names> </name><name name-style="western"><surname>Westerway</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Dietrich</surname><given-names>CF</given-names> </name></person-group><article-title>Point-of-care ultrasound in obstetrics and gynecology</article-title><source>Arch Gynecol Obstet</source><year>2021</year><month>04</month><volume>303</volume><issue>4</issue><fpage>871</fpage><lpage>876</lpage><pub-id pub-id-type="doi">10.1007/s00404-021-05972-5</pub-id><pub-id pub-id-type="medline">33558990</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shwayder</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Copel</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Stohl</surname><given-names>H</given-names> </name></person-group><article-title>Coding and legal issues in obstetric and gynecologic ultrasound</article-title><source>Obstet Gynecol Clin North Am</source><year>2019</year><month>12</month><volume>46</volume><issue>4</issue><fpage>853</fpage><lpage>862</lpage><pub-id pub-id-type="doi">10.1016/j.ogc.2019.07.012</pub-id><pub-id pub-id-type="medline">31677758</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burks</surname><given-names>K</given-names> </name><name name-style="western"><surname>Shields</surname><given-names>J</given-names> </name><name name-style="western"><surname>Evans</surname><given-names>J</given-names> </name><name name-style="western"><surname>Plumley</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gerlach</surname><given-names>J</given-names> </name><name name-style="western"><surname>Flesher</surname><given-names>S</given-names> </name></person-group><article-title>A systematic review of outpatient billing practices</article-title><source>SAGE Open Med</source><year>2022</year><volume>10</volume><fpage>20503121221099021</fpage><pub-id pub-id-type="doi">10.1177/20503121221099021</pub-id><pub-id pub-id-type="medline">35646364</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lahham</surname><given-names>S</given-names> </name><name name-style="western"><surname>Moeller</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kurzweil</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Evaluation of adherence to emergency department point-of-care ultrasound documentation and billing following intervention</article-title><source>J Med Ultrasound</source><year>2022</year><volume>30</volume><issue>3</issue><fpage>211</fpage><lpage>214</lpage><pub-id pub-id-type="doi">10.4103/jmu.jmu_76_21</pub-id><pub-id pub-id-type="medline">36484038</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lewiss</surname><given-names>RE</given-names> </name><name name-style="western"><surname>Cook</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sauler</surname><given-names>A</given-names> </name><etal/></person-group><article-title>A workflow task force affects emergency physician compliance for point-of-care ultrasound documentation and billing</article-title><source>Crit Ultrasound J</source><year>2016</year><month>12</month><volume>8</volume><issue>1</issue><fpage>5</fpage><pub-id pub-id-type="doi">10.1186/s13089-016-0041-0</pub-id><pub-id pub-id-type="medline">27207087</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Flannigan</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Adhikari</surname><given-names>S</given-names> </name></person-group><article-title>Point-of-care ultrasound work flow innovation: impact on documentation and billing</article-title><source>J Ultrasound Med</source><year>2017</year><month>12</month><volume>36</volume><issue>12</issue><fpage>2467</fpage><lpage>2474</lpage><pub-id pub-id-type="doi">10.1002/jum.14284</pub-id><pub-id pub-id-type="medline">28646595</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ng</surname><given-names>C</given-names> </name><name name-style="western"><surname>Payne</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Thomas-Mohtat</surname><given-names>R</given-names> </name><name name-style="western"><surname>Maxwell</surname><given-names>A</given-names> </name><name name-style="western"><surname>Abo</surname><given-names>A</given-names> </name></person-group><article-title>Improving point-of-care ultrasound documentation and billing accuracy in a pediatric emergency department</article-title><source>Pediatr Qual Saf</source><year>2020</year><volume>5</volume><issue>4</issue><fpage>e315</fpage><pub-id pub-id-type="doi">10.1097/pq9.0000000000000315</pub-id><pub-id pub-id-type="medline">32766490</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hall</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Hall</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gross</surname><given-names>CP</given-names> </name><etal/></person-group><article-title>Use of point-of-care ultrasound in the emergency department: insights from the 2012 Medicare national payment data set</article-title><source>J Ultrasound Med</source><year>2016</year><month>11</month><volume>35</volume><issue>11</issue><fpage>2467</fpage><lpage>2474</lpage><pub-id pub-id-type="doi">10.7863/ultra.16.01041</pub-id><pub-id pub-id-type="medline">27698180</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Ettinger</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Daum&#x00E9; III</surname><given-names>H</given-names> </name><name name-style="western"><surname>Bender</surname><given-names>EM</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Bender</surname><given-names>E</given-names> </name><name name-style="western"><surname>Daum&#x00E9; III</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ettinger</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rao</surname><given-names>S</given-names> </name></person-group><article-title>Towards linguistically generalizable NLP systems: a workshop and shared task</article-title><source>Proceedings of the First Workshop on Building Linguistically Generalizable NLP Systems</source><year>2017</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>1</fpage><lpage>10</lpage><pub-id pub-id-type="doi">10.18653/v1/W17-5401</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hossain</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rana</surname><given-names>R</given-names> </name><name name-style="western"><surname>Higgins</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Natural language processing in electronic health records in relation to healthcare decision-making: a systematic review</article-title><source>Comput Biol Med</source><year>2023</year><month>03</month><volume>155</volume><fpage>106649</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106649</pub-id><pub-id pub-id-type="medline">36805219</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shazly</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Trabuco</surname><given-names>EC</given-names> </name><name name-style="western"><surname>Ngufor</surname><given-names>CG</given-names> </name><name name-style="western"><surname>Famuyide</surname><given-names>AO</given-names> </name></person-group><article-title>Introduction to machine learning in obstetrics and gynecology</article-title><source>Obstet Gynecol</source><year>2022</year><month>04</month><day>1</day><volume>139</volume><issue>4</issue><fpage>669</fpage><lpage>679</lpage><pub-id pub-id-type="doi">10.1097/AOG.0000000000004706</pub-id><pub-id pub-id-type="medline">35272300</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Solomon</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Tabada</surname><given-names>G</given-names> </name><name name-style="western"><surname>Allen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sung</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Go</surname><given-names>AS</given-names> </name></person-group><article-title>Large-scale identification of aortic stenosis and its severity using natural language processing on electronic health records</article-title><source>Cardiovasc Digit Health J</source><year>2021</year><volume>2</volume><issue>3</issue><fpage>156</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1016/j.cvdhj.2021.03.003</pub-id><pub-id pub-id-type="medline">35265904</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zheng</surname><given-names>T</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>W</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>L</given-names> </name><etal/></person-group><article-title>A machine learning-based framework to identify type 2 diabetes through electronic health records</article-title><source>Int J Med Inform</source><year>2017</year><month>01</month><volume>97</volume><fpage>120</fpage><lpage>127</lpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2016.09.014</pub-id><pub-id pub-id-type="medline">27919371</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burns</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Mathis</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Vandervest</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Classification of current procedural terminology codes from electronic health record data using machine learning</article-title><source>Anesthesiology</source><year>2020</year><month>04</month><volume>132</volume><issue>4</issue><fpage>738</fpage><lpage>749</lpage><pub-id pub-id-type="doi">10.1097/ALN.0000000000003150</pub-id><pub-id pub-id-type="medline">32028374</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Joo</surname><given-names>H</given-names> </name><name name-style="western"><surname>Burns</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kalidaikurichi Lakshmanan</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Vydiswaran</surname><given-names>VGV</given-names> </name></person-group><article-title>Neural machine translation-based automated current procedural terminology classification system using procedure text: development and validation study</article-title><source>JMIR Form Res</source><year>2021</year><month>05</month><day>26</day><volume>5</volume><issue>5</issue><fpage>e22461</fpage><pub-id pub-id-type="doi">10.2196/22461</pub-id><pub-id pub-id-type="medline">34037526</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cersonsky</surname><given-names>TEK</given-names> </name><name name-style="western"><surname>Ayala</surname><given-names>NK</given-names> </name><name name-style="western"><surname>Pinar</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Identifying risk of stillbirth using machine learning</article-title><source>Obstet Anesth Digt</source><year>2024</year><volume>44</volume><issue>2</issue><fpage>78</fpage><lpage>79</lpage><pub-id pub-id-type="doi">10.1097/01.aoa.0001015988.43986.ab</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lokhande</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gimovsky</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sarkar</surname><given-names>I</given-names> </name></person-group><article-title>Predicting miscarriage and stillbirth using weighted ensemble machine learning [ID: 1338167]</article-title><source>Obstet Gynecol</source><year>2023</year><volume>141</volume><issue>5S</issue><fpage>28S</fpage><pub-id pub-id-type="doi">10.1097/01.AOG.0000930064.05901.88</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>W</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><etal/></person-group><article-title>BioBERT: a pre-trained biomedical language representation model for biomedical text mining</article-title><source>Bioinformatics</source><year>2020</year><month>02</month><day>15</day><volume>36</volume><issue>4</issue><fpage>1234</fpage><lpage>1240</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id><pub-id pub-id-type="medline">31501885</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><article-title>Chen S, Li Y, Lu S, Van H, Aerts HJWL, Savova GK, Bitterman DS. Correction to: Evaluating the ChatGPT family of models for biomedical reasoning and classification</article-title><source>J Am Med Inform Assoc</source><year>2024</year><volume>31</volume><issue>6</issue><fpage>1446</fpage><pub-id pub-id-type="doi">10.1093/jamia/ocae083</pub-id><pub-id pub-id-type="medline">38587877</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="web"><article-title>The Strengthening the Reporting of Observational Studies in Epidemiology (STROBE) statement: guidelines for reporting observational studies</article-title><source>EQUATOR Network</source><access-date>2024-12-18</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.equator-network.org/reporting-guidelines/strobe/">https://www.equator-network.org/reporting-guidelines/strobe/</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bird</surname><given-names>S</given-names> </name><name name-style="western"><surname>Klein</surname><given-names>E</given-names> </name><name name-style="western"><surname>Loper</surname><given-names>E</given-names> </name></person-group><source>Natural Language Processing with Python</source><year>2009</year><edition>1</edition><publisher-name>O&#x2019;Reilly Media</publisher-name><pub-id pub-id-type="other">9780596516499</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ke</surname><given-names>G</given-names> </name><name name-style="western"><surname>Meng</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Finley</surname><given-names>T</given-names> </name><etal/></person-group><article-title>LightGBM: a highly efficient gradient boosting decision tree</article-title><source>NIPS&#x2019;17: Proceedings of the 31st International Conference on Neural Information Processing Systems</source><year>2017</year><access-date>2026-07-14</access-date><fpage>3149</fpage><lpage>3157</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/10.5555/3294996.3295074">https://dl.acm.org/doi/10.5555/3294996.3295074</ext-link></comment></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Simmons</surname><given-names>A</given-names> </name><name name-style="western"><surname>Takkavatakarn</surname><given-names>K</given-names> </name><name name-style="western"><surname>McDougal</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Extracting international classification of diseases codes from clinical documentation using large language models</article-title><source>Appl Clin Inform</source><year>2025</year><month>03</month><volume>16</volume><issue>2</issue><fpage>337</fpage><lpage>344</lpage><pub-id pub-id-type="doi">10.1055/a-2491-3872</pub-id><pub-id pub-id-type="medline">39608761</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sendak</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Balu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Schulman</surname><given-names>KA</given-names> </name></person-group><article-title>Barriers to achieving economies of scale in analysis of EHR data. A cautionary tale</article-title><source>Appl Clin Inform</source><year>2017</year><month>08</month><day>9</day><volume>8</volume><issue>3</issue><fpage>826</fpage><lpage>831</lpage><pub-id pub-id-type="doi">10.4338/ACI-2017-03-CR-0046</pub-id><pub-id pub-id-type="medline">28837212</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guedalia</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lipschuetz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Novoselsky-Persky</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Real-time data analysis using a machine learning model significantly improves prediction of successful vaginal deliveries</article-title><source>Am J Obstet Gynecol</source><year>2020</year><month>09</month><volume>223</volume><issue>3</issue><fpage>437.e1-437.e15</fpage><pub-id pub-id-type="doi">10.1016/j.ajog.2020.05.025</pub-id><pub-id pub-id-type="medline">32434000</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lipschuetz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Guedalia</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rottenstreich</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Prediction of vaginal birth after cesarean deliveries using machine learning</article-title><source>Am J Obstet Gynecol</source><year>2020</year><month>06</month><volume>222</volume><issue>6</issue><fpage>613.e1-613.e12</fpage><pub-id pub-id-type="doi">10.1016/j.ajog.2019.12.267</pub-id><pub-id pub-id-type="medline">32007491</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McCoy</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Levine</surname><given-names>LD</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chivers</surname><given-names>C</given-names> </name><name name-style="western"><surname>Teel</surname><given-names>J</given-names> </name><name name-style="western"><surname>La Cava</surname><given-names>WG</given-names> </name></person-group><article-title>Intrapartum electronic fetal heart rate monitoring to predict acidemia at birth with the use of deep learning</article-title><source>Am J Obstet Gynecol</source><year>2025</year><month>01</month><volume>232</volume><issue>1</issue><fpage>116.e1-116.e9</fpage><pub-id pub-id-type="doi">10.1016/j.ajog.2024.04.022</pub-id><pub-id pub-id-type="medline">38663662</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Joly</surname><given-names>R</given-names> </name><name name-style="western"><surname>Beecy</surname><given-names>AN</given-names> </name><etal/></person-group><article-title>Implementation of a machine learning risk prediction model for postpartum depression in the electronic health records</article-title><source>AMIA Jt Summits Transl Sci Proc</source><year>2024</year><volume>2024</volume><fpage>1057</fpage><lpage>1066</lpage><pub-id pub-id-type="medline">39444417</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McAdams</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Green</surname><given-names>TL</given-names> </name></person-group><article-title>Equitable artificial intelligence in obstetrics, maternal-fetal medicine, and neonatology</article-title><source>Obstet Gynecol</source><year>2024</year><month>03</month><day>28</day><pub-id pub-id-type="doi">10.1097/AOG.0000000000005563</pub-id><pub-id pub-id-type="medline">38547488</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mathews</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Makary</surname><given-names>MA</given-names> </name></person-group><article-title>Billing quality is medical quality</article-title><source>JAMA</source><year>2020</year><month>02</month><day>4</day><volume>323</volume><issue>5</issue><fpage>409</fpage><lpage>410</lpage><pub-id pub-id-type="doi">10.1001/jama.2019.19648</pub-id><pub-id pub-id-type="medline">32016315</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mello</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Rose</surname><given-names>S</given-names> </name></person-group><article-title>Denial-artificial intelligence tools and health insurance coverage decisions</article-title><source>JAMA Health Forum</source><year>2024</year><month>03</month><day>1</day><volume>5</volume><issue>3</issue><fpage>e240622</fpage><pub-id pub-id-type="doi">10.1001/jamahealthforum.2024.0622</pub-id><pub-id pub-id-type="medline">38451493</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Ultrasound rates by clinic, model inputs, and performance over time.</p><media xlink:href="jmir_v28i1e84454_app1.docx" xlink:title="DOCX File, 428 KB"/></supplementary-material><supplementary-material id="app2"><label>Checklist 1</label><p>STROBE checklist.</p><media xlink:href="jmir_v28i1e84454_app2.docx" xlink:title="DOCX File, 34 KB"/></supplementary-material></app-group></back></article>