<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e95609</article-id><article-id pub-id-type="doi">10.2196/95609</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Demographic Confounding in Voice-Based Parkinson Disease Screening: Methodological Analysis of the Bridge2AI Voice Dataset</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Shukla</surname><given-names>Shikhar</given-names></name><degrees>BDS, MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Naliyatthaliyazchayil</surname><given-names>Parvati</given-names></name><degrees>PharmD, MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gichoya</surname><given-names>Judy W</given-names></name><degrees>MD, MS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Purkayastha</surname><given-names>Saptarshi</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Biomedical Informatics and Engineering, Luddy School of Informatics, Computing, and Engineering, Indiana University</institution><addr-line>535 W Michigan St., IT 475J</addr-line><addr-line>Indianapolis</addr-line><addr-line>IN</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Radiology and Imaging Sciences, Emory University School of Medicine, Emory University</institution><addr-line>Atlanta</addr-line><addr-line>GA</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Kotting</surname><given-names>Carsten</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Boutsen</surname><given-names>Frank R</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Saptarshi Purkayastha, PhD, Department of Biomedical Informatics and Engineering, Luddy School of Informatics, Computing, and Engineering, Indiana University, 535 W Michigan St., IT 475J, Indianapolis, IN, 46202, United States, 1 3172740439; <email>saptpurk@iu.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>20</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e95609</elocation-id><history><date date-type="received"><day>18</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>02</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>06</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Shikhar Shukla, Parvati Naliyatthaliyazchayil, Judy W Gichoya, Saptarshi Purkayastha. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 20.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e95609"/><abstract><sec><title>Background</title><p>Voice-based deep learning models for Parkinson disease (PD) and dementia screening report areas under the curve (AUCs) of 0.85&#x2010;0.97, but rarely audit demographic confounding. Because speech changes substantially with age, case-control age imbalance alone can produce high classification performance independent of disease.</p></sec><sec><title>Objective</title><p>We audited the Bridge to Artificial Intelligence (Bridge2AI) Voice Dataset v3.0.0 with three objectives: (1) quantify age and site confounding in voice screening for PD and dementia, (2) evaluate whether a disease-specific acoustic signal persists after demographic adjustment, and (3) propose minimum reporting standards.</p></sec><sec sec-type="methods"><title>Methods</title><p>We fine-tuned an audio spectrogram transformer (AST; 86.4 million parameters) using 5-fold participant-level cross-validation for PD (n=253) and dementia (n=221). Logistic regression on age and sex provided a demographic-only baseline under identical splits. Confounding was assessed by (1) restricting evaluation to ages 60&#x2010;80 years, (2) restricting to US participants because all Canadian PD (n=62) and all Canadian dementia (n=70) participants were cases with no Canadian controls, (3) retraining AST from scratch on the age-restricted subgroup, (4) 1:1 nearest-neighbor propensity-score matching on age and sex in the US age-restricted PD subgroup, and (5) applying v3.0.0-trained models to v2.0.1 spectrograms of the same participants.</p></sec><sec sec-type="results"><title>Results</title><p>On the full cohort, age alone was statistically indistinguishable from the AST (PD: AUC 0.875 vs 0.843, DeLong <italic>P</italic>=.36; dementia: 0.905 vs 0.895, <italic>P</italic>=.75), indicating a demographic shortcut; because both cohorts share the same 148-person control group, this parallel pattern is one dataset-level artifact, not 2 independent confirmations. Within ages 60&#x2010;80 years, the AST exceeded age-only for both conditions (PD: 0.787 vs 0.568, &#x0394;AUC +0.225, 95% CI +0.089 to +0.361; <italic>P</italic>=.001, n=129; dementia: 0.809 vs 0.598, &#x0394;AUC +0.215, 95% CI +0.056 to +0.375; <italic>P</italic>=.008, n=95), though the dementia result reflects residual site confounding. Three convergent estimates of PD-specific signal, post hoc age-restricted (0.787), de novo retrained (0.798), and propensity-matched (US, 28 pairs; 0.760; <italic>P</italic>=.02 vs age-only), all fell substantially below the 0.843 full-cohort AUC. Under the most conservative adjustment (US-only, ages 60&#x2010;80 years), AST AUC was 0.726, which we consider the least confounded estimate. Cross-version preprocessing testing produced AUC 0.618, indicating brittle representations. Of 13 published voice-PD studies, only 7 reported case and control age distributions, and none reported a demographic-only baseline or an age-restricted evaluation.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Demographic confounding dominates full-cohort voice-screening metrics on the Bridge2AI Voice Dataset for both PD and dementia. After age restriction, site control, and propensity matching, residual AST performance for PD (AUC 0.726&#x2010;0.798) remained significantly above demographic baselines, supporting a disease-specific acoustic signal substantially weaker than full-cohort metrics suggest; for dementia, insufficient US cases precluded similar estimates. Voice biomarker studies should report case and control demographic distributions, demographic-only baselines under identical splits, and age-restricted performance alongside full-cohort metrics.</p></sec></abstract><kwd-group><kwd>confounding factors</kwd><kwd>voice biomarkers</kwd><kwd>Parkinson disease</kwd><kwd>age factors</kwd><kwd>audio spectrogram transformer</kwd><kwd>Bridge2AI</kwd><kwd>speech acoustics</kwd><kwd>digital health</kwd><kwd>validation studies</kwd><kwd>dementia</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Voice-based screening for neurological disorders has emerged as a promising noninvasive approach, with deep learning models reporting increasingly high classification accuracy for conditions including Parkinson disease (PD) [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Speech reflects the coordinated activity of motor, cognitive, and affective neural systems [<xref ref-type="bibr" rid="ref4">4</xref>], and subtle changes in voice quality, articulation, and prosody have been associated with many health conditions [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Recent studies applying audio spectrogram transformers (ASTs) [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>], self-supervised models [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>], and convolutional architectures [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>] to PD classification have reported areas under the curve (AUCs) ranging from 0.85 to 0.97, fueling optimism about clinical translation. The AST [<xref ref-type="bibr" rid="ref7">7</xref>] is a vision transformer adapted for audio classification, pretrained on AudioSet (2 million clips and 527 classes). It has achieved state-of-the-art results on audio classification benchmarks and has been applied to pathological voice detection [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>However, these performance metrics may be systematically inflated by a confounder that is rarely examined: age. PD is predominantly a disease of aging, with onset typically occurring at approximately 60 years of age [<xref ref-type="bibr" rid="ref13">13</xref>]. At the same time, the human voice undergoes substantial age-related changes, including decreased pitch, increased jitter and shimmer, reduced harmonic-to-noise ratio, and altered vocal fold dynamics [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>], that are independent of any neurological pathology. When PD cohorts are constructed by comparing older patients against younger healthy volunteers, as is common in convenience-sampled datasets, any classifier (including one with no access to speech at all) may achieve high discrimination simply by exploiting the age gap. This problem is well recognized in medical imaging [<xref ref-type="bibr" rid="ref16">16</xref>] but has received remarkably little attention in the voice biomarker literature.</p><p>The absence of demographic confounding analysis in voice-based PD screening represents a critical methodological gap [<xref ref-type="bibr" rid="ref17">17</xref>]. While Brenner et al [<xref ref-type="bibr" rid="ref18">18</xref>] demonstrated that age-based selection bias in the mPower dataset inflates PD classification scores, and Ozbolt et al [<xref ref-type="bibr" rid="ref19">19</xref>] identified age imbalance between groups as a key methodological concern in sustained-vowel analyses, no study has systematically quantified demographic confounding across multiple conditions on a modern multisite dataset or proposed minimum reporting standards. Most published studies report overall classification metrics without controlling for, or even reporting, the age distribution of cases and controls. Studies that do report demographics often show substantial age imbalances [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref20">20</xref>] but do not assess their impact on performance; for example, Rahman et al [<xref ref-type="bibr" rid="ref21">21</xref>] reported an 8-year age gap between PD and controls and performed age-trimming (excluding participants younger than 50 years) but did not test a demographic-only baseline or perform age-restricted evaluation (Table S12 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>The severity of this gap is underscored by how readily the confounding can be detected. This study originated from the CIMDAR-HIVE (Developing a Hive Learning and Datathon Supported Course on Imaging and Multimodal Data for Resource-Limited Institutions) training program under the National Institutes of Health (NIH) Common Fund Data Ecosystem initiative, where the first and senior authors designed bias investigation exercises for the 2025 Emory Health AI Summer School &#x0026; Datathon [<xref ref-type="bibr" rid="ref22">22</xref>]. During these sessions, participants with no prior experience in voice biomarker research identified age confounding in the Bridge to Artificial Intelligence (Bridge2AI) dataset within hours, a finding that, despite its simplicity, has not been reported or addressed in any published study using this corpus. The formal analysis presented here systematically investigates that initial observation.</p><p>This study addresses this gap directly using the Bridge2AI Voice Dataset v3.0.0 [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>], a publicly available corpus of 833 participants across 5 clinical sites. We fine-tuned an AST [<xref ref-type="bibr" rid="ref7">7</xref>] for PD and dementia screening under strict participant-level cross-validation, and then systematically investigated the role of demographic confounding through age-only baselines, age-restricted subgroup evaluation, site-level analysis, and cross-version preprocessing robustness testing. While demonstrated on a single dataset, the demographic structure we identify is common to many voice datasets; we argue that age-restricted subgroup evaluation should become a standard reporting requirement for voice-based disease screening research.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>We developed a transformer-based speech screening framework (<xref ref-type="fig" rid="figure1">Figure 1</xref>) using spectrogram representations and a pretrained AST, applied to the publicly available Bridge2AI Voice Dataset v3.0.0. The same architecture, preprocessing, training procedure, and evaluation protocol were applied to both PD and dementia; only the resulting cohort compositions differed (<xref ref-type="table" rid="table1">Table 1</xref>). Beyond the confounding analysis, we evaluate the AST for dementia screening, compare spectrogram-based and phonetic posteriorgram (PPG) [<xref ref-type="bibr" rid="ref25">25</xref>] representations, assess baseline models, and examine the effect of speech task selection on screening performance. Depression screening (44 cases) was conducted as an exploratory analysis reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Methodology overview of the demographic confounding analysis framework. AST: audio spectrogram transformer; AUC: area under the curve; Bridge2AI: Bridge to Artificial Intelligence; CV: cross-validation; GELU: Gaussian Error Linear Unit; OOF: out-of-fold; PD: Parkinson disease.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95609_fig01.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Selected speech tasks for PD<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> and dementia screening (Bridge2AI<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> voice dataset v3.0.0; 833 participants; 5 clinical sites)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Condition</td><td align="left" valign="bottom">Tasks</td><td align="left" valign="bottom">Participants (cases and controls)</td><td align="left" valign="bottom">Recordings</td><td align="left" valign="bottom">Case (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Parkinson disease</td><td align="left" valign="top">8<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">253 (106, 147)</td><td align="left" valign="top">2131</td><td align="left" valign="top">41.9</td></tr><tr><td align="left" valign="top">Dementia</td><td align="left" valign="top">8<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">221 (73, 148)</td><td align="left" valign="top">1791</td><td align="left" valign="top">33</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>PD: Parkinson disease.</p></fn><fn id="table1fn2"><p><sup>b</sup>Bridge2AI: Bridge to Artificial Intelligence.</p></fn><fn id="table1fn3"><p><sup>c</sup>Prolonged vowel; glides (high-to-low, low-to-high); diadochokinesis (pataka); rainbow passage; picture description; story recall; maximum phonation time.</p></fn><fn id="table1fn4"><p><sup>d</sup> Eight tasks spanning phonation, articulation, and cognitive-linguistic domains were selected based on case and control representation.</p></fn></table-wrap-foot></table-wrap><p>To assess demographic confounding, we additionally trained a logistic regression classifier using only age and sex as features, with identical cross-validation splits. Age-restricted subgroup analyses were performed both post hoc (evaluating existing out-of-fold [OOF] predictions on age-restricted subsets) and prospectively (retraining the AST from scratch on the age-restricted subgroup). Country-level analysis was conducted as a proxy for multisite evaluation, and attention map analysis was performed to characterize the spectrotemporal features driving classification. Full details of all supporting analyses are provided below.</p></sec><sec id="s2-2"><title>Dataset Description</title><p>This study used the Bridge2AI Voice Dataset v3.0.0 [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>], a publicly available corpus comprising 833 participants with voice recordings from individuals with neurological and psychiatric conditions as well as healthy controls, collected across 5 clinical sites. The dataset includes multiple structured speech tasks designed to elicit a broad range of vocal and speech characteristics, such as sustained phonation, pitch control, articulatory agility, connected speech, and cognitive-linguistic tasks [<xref ref-type="bibr" rid="ref24">24</xref>].</p><p>Voice recordings were provided in the form of log-Mel spectrograms, stored in Parquet format (<italic>features/torchaudio_mel_spectrogram.parquet</italic>). Each spectrogram consisted of 60 Mel frequency bins and a variable number of time frames, reflecting differences in recording duration across tasks and participants. Additionally, v3.0.0 provides PPGs extracted from a pretrained neural speech model [<xref ref-type="bibr" rid="ref25">25</xref>], stored in the ppgs.parquet file in the features folder, with 40 phoneme classes and variable temporal length.</p><p>Participant-level phenotype data were obtained from the hierarchical phenotype annotations under the diagnosis subdirectory of the phenotype directory. PD cases (106 participants) were identified from <italic>parkinsons_disease.tsv</italic>, dementia cases (73 participants) from <italic>cognitive_impairment.tsv</italic>, and controls (148 participants) from <italic>control.tsv</italic>. One participant appearing in both PD and control files was assigned to the PD group. Spectrograms and PPGs were merged with phenotype data using participant identifiers (6-digit numeric, zero-padded).</p><p>For cross-version preprocessing robustness testing, Bridge2AI Voice Dataset v2.0.1 was used. v2.0.1 comprises 442 participants with spectrograms of 201 Mel frequency bins, a single phenotype file, and 8-character hexadecimal participant identifiers. Although the identifier formats differ between versions, session-level identifier matching confirms that 441 of 442 v2.0.1 participants reappear in v3.0.0 under reanonymized numeric identifiers, consistent with the PhysioNet release notes describing v3.0.0 as adding &#x201C;an additional 391 participants&#x201D; to the prior release. This cumulative-release convention is standard practice for biomedical reference cohorts (eg, UK Biobank, ADNI, and MIMIC), where successive versions extend rather than replace prior data; researchers comparing across Bridge2AI Voice releases should not interpret v2.0.1 and v3.0.0 as independent cohorts. The two versions therefore represent the same participant cohort with different spectrogram extraction pipelines, enabling a controlled assessment of preprocessing sensitivity.</p></sec><sec id="s2-3"><title>Task Selection and Inclusion Criteria</title><p>To support condition-specific screening while maintaining a unified modeling pipeline, speech tasks were selected based on adequate representation of both cases and controls within the v3.0.0 dataset. Tasks were prioritized according to the number of unique case and control participants with available recordings, ensuring sufficient representation for stable participant-level model training and evaluation.</p><p>Eight tasks with adequate PD and control representation were selected: prolonged vowel, glides (high-to-low and low-to-high), diadochokinesis (pataka), rainbow passage, picture description, story recall, and maximum phonation time. These tasks span sustained phonation, pitch control, articulatory agility, connected speech, and cognitive-linguistic domains. The same 8-task set was used for both PD and dementia analyses. Per-task recording distributions are provided in Table S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Depression screening used a separate task set selected for case coverage (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>Only spectrograms with at least 100 timeframes were retained to ensure sufficient spectrotemporal context for modeling. <xref ref-type="table" rid="table1">Table 1</xref> summarizes the selected task sets for each condition.</p></sec><sec id="s2-4"><title>Preprocessing</title><sec id="s2-4-1"><title>Overview</title><p>Raw spectrograms (60 Mel bins&#x00D7;variable time frames in v3.0.0; 201 Mel bins in v2.0.1 used for preprocessing robustness testing) were preprocessed to produce fixed-size inputs compatible with the AST architecture. Spectrograms with fewer than 100 timeframes were excluded. This filter operates at the recording level rather than the participant level, so it removes short recordings rather than entire participants; its effect is therefore a subset of the broader task-missingness mechanism discussed below. Temporal length was standardized to 1024 frames using reflect padding (for shorter recordings) or center cropping (for longer ones). The frequency axis was resized to 128 Mel bins via linear interpolation, yielding a uniform 128&#x00D7;1024 representation. Fold-specific z-score normalization was applied, with statistics computed exclusively from each fold&#x2019;s training partition to prevent information leakage. Representative preprocessed spectrograms are shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>. Full preprocessing details, including the rationale for reflect padding over zero padding and the interpolation method, are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Spectrogram preprocessing pipeline for audio spectrogram transformer input. Example Mel spectrogram from a Parkinson disease participant illustrating sequential preprocessing steps: (1) raw spectrogram with variable temporal length (F &#x00D7; T, where <italic>F</italic>=60 in v3.0.0), (2) center crop or reflective padding to a fixed length of 1024 time frames (F &#x00D7; 1024), (3) interpolation-based resizing to 128 Mel bins (128 &#x00D7;1024) to match audio spectrogram transformer input requirements, and (4) <italic>z</italic>-score normalization (mean approximately 0, SD approximately 1).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95609_fig02.png"/></fig></sec><sec id="s2-4-2"><title>PPG Preprocessing</title><p>PPGs (40 phoneme classes&#x00D7;variable time) were preprocessed identically to spectrograms (reflect-padded or center-cropped to 1024 frames, resized to 128&#x00D7;1024, fold-normalized). We note that interpolating 40-dimensional PPG vectors to 128 dimensions via linear interpolation introduces artifacts, as PPGs represent posterior probabilities over discrete phoneme classes. The interpolated values lack meaningful phonetic interpretation. This preprocessing decision limits the validity of comparisons between PPG-based and spectrogram-based AST performance. A supplementary 1-dimensional (1D) convolutional neural network (CNN) evaluated on native 40-dimensional PPGs is described in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s2-5"><title>Cross-Version Preprocessing Robustness</title><p>To assess the sensitivity of learned representations to preprocessing changes, the 5-fold models trained on v3.0.0 spectrograms were applied as an ensemble (averaged predicted probabilities) to the v2.0.1 spectrograms of the same participants (442 participants). Because v2.0.1 participants are a subset of v3.0.0 (confirmed via session identifier matching; see the &#x201C;Dataset Description&#x201D; section), this test isolates the effect of the feature extraction pipeline (60 vs 201 Mel bins) from population-level differences. Because v3.0.0 spectrograms (60 Mel bins) are upsampled to 128 bins via linear interpolation, while v2.0.1 spectrograms (201 Mel bins) are downsampled to 128 bins, this test conflates preprocessing domain shift with any genuine generalization gap. The resulting AUC should be interpreted as a measure of model sensitivity to the feature extraction pipeline, not as a pure external validation. v2.0.1 spectrograms were resized to 128&#x00D7;1024 and normalized using each fold&#x2019;s v3.0.0 training statistics. The full protocol is described in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-6"><title>Participant-Level Data Splitting</title><p>All spectrograms from a given participant were assigned exclusively to training or validation, with no participant overlap between partitions. A preliminary 80/20 split was used during pipeline development; all reported results are based on the 5-fold cross-validation described below.</p></sec><sec id="s2-7"><title>Model Architecture and Training</title><p>The classifier was built on the AST [<xref ref-type="bibr" rid="ref7">7</xref>], a vision transformer adapted for audio, initialized from AudioSet-pretrained weights (MIT/ast-finetuned-audioset-10-10-0.4593 via Hugging Face). Input spectrograms (128&#x00D7;1024) were divided into nonoverlapping patches with learned positional embeddings, and the full transformer backbone (approximately 86.4 million parameters) was fine-tuned end-to-end. A classification head consisting of layer normalization, a 256-unit dense layer with Gaussian Error Linear Unit (GELU) activation, dropout (rate=0.3), and a 2-class output layer was appended to the pooled transformer output.</p><p>SpecAugment-style time and frequency masking was applied during training (50% probability each). Models were trained using AdamW with differential learning rates (backbone 5&#x00D7;10&#x207B;&#x2076;; head 5&#x00D7;10&#x207B;&#x2074;), cosine annealing, focal loss with per-fold inverse class-frequency weights, and gradient clipping. Early stopping was based on a composite of participant-level AUC (weight 0.4) and optimized <italic>F</italic><sub>1</sub>-score (weight 0.6), evaluated on the held-out validation fold after each epoch, with training terminating after 10 epochs without improvement. Because the epoch-selection criterion was computed on the same held-out fold whose predictions were subsequently aggregated as OOF performance, the reported cross-validated metrics carry an optimization bias; we quantify the direction of this bias in the &#x201C;Limitations&#x201D; section. The higher <italic>F</italic><sub>1</sub>-score weight reflects the clinical priority of balanced sensitivity-specificity over discrimination alone. Full hyperparameter values are provided in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-8"><title>Evaluation</title><sec id="s2-8-1"><title>Participant-Level Evaluation</title><p>For each condition, predictions were aggregated at the participant level by averaging the predicted probabilities across all recordings from the same participant. Participant-level predictions were computed as unweighted means of recording-level probabilities across all tasks completed by that individual. Because not all participants completed all tasks (recording counts range from 247 to 290 across the 8 selected tasks), this averaging implicitly weights toward the tasks present for each participant. If task completion correlates with age or condition, for example, if older or more impaired participants systematically fail to complete cognitively demanding tasks, the aggregated predictions may partially reflect missingness patterns rather than acoustic features. Performance was evaluated using AUC-ROC (area under the receiver operating characteristic curve; threshold-independent) as the primary metric, supplemented by <italic>F</italic><sub>1</sub>-score, precision, and recall. To prevent test-set leakage in threshold selection for the secondary metrics, optimal thresholds were determined per fold using a leave-one-fold-out (LOFO) Youden J procedure: for each fold k, the threshold was computed from the OOF predictions of the remaining 4 folds and applied to fold-k validation predictions. This ensures that the threshold for any participant&#x2019;s classification is independent of that participant&#x2019;s prediction. For comparison, results at a fixed threshold of 0.5 and at training partition&#x2013;derived thresholds are reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Pairwise differences in AUC between models evaluated on the same participants were tested using the DeLong test, which yields exact 95% CIs for the AUC difference and accounts for the correlation induced by paired predictions. Because 9 AST-versus-demographic-baseline comparisons were performed across the full and age-restricted cohorts for both conditions, <italic>P</italic> values were adjusted for multiple comparisons using the Benjamini-Hochberg false discovery rate (FDR) procedure; both raw <italic>P</italic> values and adjusted q-values are reported. This ensures that evaluation reflects participant-level classification rather than individual recording-level performance.</p></sec><sec id="s2-8-2"><title>Five-Fold Cross-Validation</title><p>To assess model robustness for each condition, 5-fold stratified cross-validation was performed at the participant level using a fixed random seed (random state=42), ensuring that each participant contributed to validation in exactly 1-fold. Cross-validation was performed over all participants in each condition-specific cohort, not restricted to the training partition described in the &#x201C;Participant-Level Data Splitting&#x201D; section. Within each fold, a fresh AST model was initialized from the same pretrained weights and trained with the procedure described above. Fold-specific normalization statistics were computed from the training partition of each fold and applied to the corresponding validation partition. Participant-level AUC-ROC, <italic>F</italic><sub>1</sub>-score, recall, and precision were computed for each fold, and OOF predictions were aggregated across all 5 folds to report true cross-validated performance. Mean and SD of per-fold metrics are reported. Bootstrap 95% CIs (2000 iterations with replacement) were computed from the OOF predictions for key AUC estimates.</p></sec><sec id="s2-8-3"><title>Demographic Confounding Analysis</title><p>To assess demographic confounding, participant-level age and sex at birth were extracted from the demographics file and used to train a logistic regression classifier with balanced class weights. The metadata logistic regression used the same 5-fold cross-validation splits as the AST, applied independently to both the PD cohort (252/253 participants with complete data) and dementia cohort (220/221 participants). OOF predictions from the metadata-only model were compared to the AST&#x2019;s OOF predictions to quantify the contribution of demographic covariates.</p><p>To disentangle age-related voice changes from disease-specific signals, the AST&#x2019;s existing OOF predictions were evaluated on age-restricted subgroups (55 to 85 and 60 to 80 years) for both conditions and compared against age-only and metadata-only AUC within each subgroup. For PD, AST-derived and metadata-based probabilities were additionally combined using equal-weight (0.5/0.5) soft voting with identical fold assignments.</p></sec><sec id="s2-8-4"><title>Age-Restricted Retraining</title><p>To validate the post hoc age-restricted estimate, the AST was retrained from scratch using only participants aged 60 to 80 years (129 participants: 84 PD, 45 controls; PD: mean age 71.4, SD 5.8 years vs control: mean age 70.0, SD 6.5 years). The same architecture, hyperparameters, and 5-fold stratified cross-validation protocol were used. An age-only logistic regression was trained on the same cohort for comparison.</p></sec><sec id="s2-8-5"><title>Country-Level Analysis</title><p>In the absence of explicit site identifiers in v3.0.0, participant country of origin (United States or Canada, from <italic>demographics.tsv</italic>) was used as a coarse proxy for institutional variation. The AST&#x2019;s existing OOF predictions were stratified by country, and AUC-ROC was computed for each country subset. Age-restricted evaluation (ages 60 to 80 years) was performed within the US subset.</p></sec><sec id="s2-8-6"><title>Attention Map Analysis</title><p>Classify token (CLS-token) attention weights were extracted from all 12 transformer layers of the fold-1 AST model for prolonged-vowel recordings (105 PD and 147 controls). Statistical testing included permutation tests, FDR-corrected pixel-wise comparisons, Cohen <italic>d</italic> effect sizes, and frequency-band analysis. Full methodology is described in Section 7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s2-9"><title>Sensitivity and Supporting Analyses</title><p>A sensitivity analysis using all available speech tasks (rather than the 8 selected tasks) was conducted with the same AST architecture and evaluation protocol, with minor hyperparameter adjustments for the larger dataset (max epochs 20, patience 5, and constant learning rate). A foundation model comparison evaluated PPGs [<xref ref-type="bibr" rid="ref25">25</xref>] processed through the same AST architecture, and a 1D CNN on native 40-dimensional PPGs, using identical cross-validation splits.</p><p>This study is reported in accordance with the TRIPOD+AI (Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis&#x2013;Artificial Intelligence) guidelines for prediction model studies [<xref ref-type="bibr" rid="ref16">16</xref>]. The completed checklist is provided in <xref ref-type="supplementary-material" rid="app2">Checklist 1</xref>.</p></sec><sec id="s2-10"><title>Ethical Considerations</title><p>The Bridge2AI Voice Dataset v3.0.0 (primary) and v2.0.1 (preprocessing robustness testing) used in this study are publicly available through PhysioNet. Access requires credentialing through PhysioNet and execution of the Bridge2AI Voice Registered Access Data Use Agreement. Data collection and sharing were approved by the University of South Florida Institutional Review Board. As this study exclusively used deidentified, publicly available datasets accessed under a data use agreement, no additional institutional review board approval was required. Code for this analysis is available on GitHub [<xref ref-type="bibr" rid="ref26">26</xref>].</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Cohort Characteristics and Demographic Imbalance</title><p>Condition-specific cohorts were constructed from the hierarchical phenotype annotations in Bridge2AI v3.0.0 (<xref ref-type="fig" rid="figure3">Figure 3</xref>). PD cases (106 participants) were identified from <italic>parkinsons_disease.tsv</italic> and dementia cases (73 participants) from <italic>cognitive_impairment.tsv</italic>. Controls (147 participants after removing 1 participant with both PD and control labels) were drawn from <italic>control.tsv</italic>. Comorbidity across conditions was minimal (1 participant with both PD and dementia; Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), supporting the interpretation of condition-specific analyses. Eight speech tasks spanning sustained phonation, pitch control, articulatory agility, connected speech, and cognitive-linguistic domains were selected. Participant-level characteristics are summarized in <xref ref-type="table" rid="table2">Table 2</xref>. All results reported below are at the participant level, with predicted probabilities aggregated across recordings per individual. Depression screening (44 cases) is reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> due to the small sample size.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Participant flow diagram for condition-specific cohort construction from Bridge2AI Voice Dataset v3.0.0 (833 participants; 5 US and Canadian clinical sites). AST: audio spectrogram transformer; AUC: area under the curve; Bridge2AI: Bridge to Artificial Intelligence; OOF: out-of-fold; PD: Parkinson disease.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95609_fig03.png"/></fig><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Cohort composition and demographics for PD<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> and dementia screening (Bridge2AI<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> v3.0.0; 5 US and Canadian clinical sites)<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Fields</td><td align="left" valign="bottom">Parkinson disease</td><td align="left" valign="bottom">Dementia</td></tr></thead><tbody><tr><td align="left" valign="top">Participants</td><td align="left" valign="top">253</td><td align="left" valign="top">221</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cases and controls</td><td align="left" valign="top">106, 147</td><td align="left" valign="top">73<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup>, 148</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Recordings</td><td align="left" valign="top">2131</td><td align="left" valign="top">1791</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Selected tasks</td><td align="left" valign="top">8</td><td align="left" valign="top">8</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Case proportion (%)</td><td align="left" valign="top">41.9</td><td align="left" valign="top">33</td></tr><tr><td align="left" valign="top" colspan="3">Demographics</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Case age (years), mean (SD)</td><td align="left" valign="top">72.3 (8.7)</td><td align="left" valign="top">74.9 (7.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Control age (years), mean (SD)</td><td align="left" valign="top">44.8 (19.3)</td><td align="left" valign="top">44.7 (19.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age gap (years)</td><td align="left" valign="top">27.5</td><td align="left" valign="top">30.2</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Case male, n (%)</td><td align="left" valign="top">60 (56.6)</td><td align="left" valign="top">45 (61.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Control male, n (%)</td><td align="left" valign="top">70 (47.6)</td><td align="left" valign="top">71 (48.0)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>PD: Parkinson disease.</p></fn><fn id="table2fn2"><p><sup>b</sup>Bridge2AI: Bridge to Artificial Intelligence.</p></fn><fn id="table2fn3"><p><sup>c</sup> Both cohorts share the same control group (n=147&#x2010;148, mean age 44.7, SD 19.3 to mean 44.8, SD 19.3 years), producing case-control age gaps of 27.5 years (PD) and 30.2 years (dementia).</p></fn><fn id="table2fn4"><p><sup>d</sup>One participant was excluded due to a comorbid Parkinson disease diagnosis.</p></fn></table-wrap-foot></table-wrap><p>Both cohorts exhibit striking demographic imbalances (<xref ref-type="table" rid="table2">Table 2</xref>). PD cases are on average 27.5 years older than controls (mean 72.3, SD 8.7 years vs mean 44.8, SD 19.3 years), and the dementia gap is even larger at 30.2 (mean 74.9, SD 7.6 vs mean 44.7, SD 19.3) years. This shared pattern arises because both conditions predominantly affect older adults, while the control group, common to both cohorts, skews young (mean 44.7, SD 19.3 years to mean 44.8, SD 19.3 years). Sex distributions are more balanced (56.6%&#x2010;61.6% vs 47.6%&#x2010;48%, male). This demographic structure is not unique to Bridge2AI; similar or unreported age imbalances characterize many voice-based disease screening datasets. The consequences of this imbalance for model performance are the central focus of this study.</p></sec><sec id="s3-2"><title>Age Confounding Dominates Screening Performance</title><p>To quantify the impact of the age imbalance on screening performance, we compared the fine-tuned AST against demographic-only classifiers using the same 5-fold cross-validation splits. The analysis was performed for both PD (252 participants with complete demographic data) and dementia (220 participants).</p></sec><sec id="s3-3"><title>PD</title><sec id="s3-3-1"><title>Age Alone Outperforms the AST</title><p>A logistic regression classifier using only participant age achieved an OOF AUC of 0.875 for PD classification, exceeding the fine-tuned AST (AUC 0.843), although DeLong test indicates the difference is not statistically significant (&#x0394;AUC=&#x2013;0.028, 95% CI &#x2013;0.088 to +0.032; <italic>P</italic>=.36; q=0.462). Adding sex at birth provided no additional benefit (age+sex AUC 0.869; sex-only AUC 0.543; logistic regression coefficient approximately 0.1 for sex vs approximately 2.1 for age). This result demonstrates that the age gap between cases and controls is the primary source of discriminability in the full PD cohort: a model with no access to speech recordings outperforms an 86.4-million-parameter deep learning model trained on spectrograms.</p></sec><sec id="s3-3-2"><title>Age-Restricted Subgroup Analysis Isolates Residual Disease Signal</title><p>To disentangle age-related voice changes from disease-specific features, we evaluated the AST&#x2019;s existing OOF predictions on age-restricted subgroups where the age gap was minimized. Restricting to participants aged 55 to 85 years (151 participants: 102 PD, 49 controls; PD: mean age 73.0, SD 7.1 years vs control: mean age 68.9, SD 7.2 years) reduced age-only AUC from 0.875 to 0.655, while the AST maintained an AUC of 0.805. In the narrower 60 to 80 window (129 participants: 84 PD, 45 controls; PD: mean age 71.4, SD 5.8 years vs control: mean age 70.0, SD 6.5 years), age-only AUC dropped to 0.568 (near chance), while the AST retained an AUC of 0.787 (<xref ref-type="table" rid="table3">Table 3</xref>).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Demographic confounding analysis for PD<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> screening (Bridge2AI<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> v3.0.0; 252 participants with complete demographics; 5-fold stratified cross-validation, participant-level, out-of-fold)<sup><xref ref-type="table-fn" rid="table3fn1">c,d,e,f,g</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Cohort</td><td align="left" valign="bottom">N</td><td align="left" valign="bottom">AUC-ROC<sup><xref ref-type="table-fn" rid="table3fn8">h</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Full cohort (age gap=27.5 years<italic>)</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age only (raw)</td><td align="left" valign="top">All</td><td align="left" valign="top">252</td><td align="left" valign="top">0.875</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Metadata only (age+sex)</td><td align="left" valign="top">All</td><td align="left" valign="top">252</td><td align="left" valign="top">0.869</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AST<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup> (spectrogram)</td><td align="left" valign="top">All</td><td align="left" valign="top">252</td><td align="left" valign="top">0.843</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AST+Metadata (soft voting)</td><td align="left" valign="top">All</td><td align="left" valign="top">252</td><td align="left" valign="top">0.924</td></tr><tr><td align="left" valign="top" colspan="4">Age-restricted subgroup (age gap=4.1 years)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AST (spectrogram)</td><td align="left" valign="top">Ages 55&#x2010;85</td><td align="left" valign="top">151</td><td align="left" valign="top">0.805</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age only (raw)</td><td align="left" valign="top">Ages 55&#x2010;85</td><td align="left" valign="top">151</td><td align="left" valign="top">0.655</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Metadata only (age+sex)</td><td align="left" valign="top">Ages 55&#x2010;85</td><td align="left" valign="top">151</td><td align="left" valign="top">0.641</td></tr><tr><td align="left" valign="top" colspan="4">Strictly age-restricted subgroup (age gap=1.4 years<italic>)</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AST (spectrogram)</td><td align="left" valign="top">Ages 60&#x2010;80</td><td align="left" valign="top">129</td><td align="left" valign="top">0.787</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age only (raw)</td><td align="left" valign="top">Ages 60&#x2010;80</td><td align="left" valign="top">129</td><td align="left" valign="top">0.568</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Metadata only (age+sex)</td><td align="left" valign="top">Ages 60&#x2010;80</td><td align="left" valign="top">129</td><td align="left" valign="top">0.552</td></tr><tr><td align="left" valign="top" colspan="4">Age-restricted retraining (trained exclusively on 60&#x2010;80 subgroup)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AST retrained (spectrogram)</td><td align="left" valign="top">Ages 60&#x2010;80</td><td align="left" valign="top">129</td><td align="left" valign="top">0.798</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age only (retrained cohort)</td><td align="left" valign="top">Ages 60&#x2010;80</td><td align="left" valign="top">129</td><td align="left" valign="top">0.555</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>PD: Parkinson disease.</p></fn><fn id="table3fn2"><p><sup>b</sup>Bridge2AI: Bridge to Artificial Intelligence.</p></fn><fn id="table3fn3"><p><sup>c</sup>Bootstrap 95% CIs (2000 iterations): AST full cohort (0.791-0.890).</p></fn><fn id="table3fn4"><p><sup>d</sup>AST age-restricted 60&#x2010;80 (0.695-0.863).</p></fn><fn id="table3fn5"><p><sup>e</sup>AST retrained (0.710, 0.878).</p></fn><fn id="table3fn6"><p><sup>f</sup>Age-only age-restricted (0.458, 0.667).</p></fn><fn id="table3fn7"><p><sup>g</sup>Age-only AUC (0.875) exceeds the AST (0.843) on the full cohort; under age restriction (60&#x2010;80 years, n=129), age-only drops to 0.568 while the AST retains 0.787. Age-restricted retraining (bottom) yields 0.798.</p></fn><fn id="table3fn8"><p><sup>h</sup>AUC-ROC: area under the receiver operating characteristic curve. </p></fn><fn id="table3fn9"><p><sup>i</sup>AST: audio spectrogram transformer.</p></fn></table-wrap-foot></table-wrap><p>This age-restricted AUC of 0.787 (bootstrap 95% CI 0.695-0.863) represents the most credible estimate of the AST&#x2019;s disease-specific discriminative capacity, as it is obtained under conditions where demographic confounding can no longer explain the result.</p><p>The pattern across progressively narrower age windows is consistent and interpretable: as the age gap narrows, the age-only classifier approaches chance performance, while the AST retains substantial discrimination. This demonstrates that the AST learns genuine disease-relevant acoustic features, but that these features account for an AUC of approximately 0.787, not the 0.843 reported on the full cohort. The AUC difference of 0.056 is attributable to age confounding. Late fusion of AST and metadata probabilities through equal-weight soft voting yielded a combined AUC of 0.924 on the full cohort, confirming that the AST captures complementary information beyond demographics.</p></sec><sec id="s3-3-3"><title>Age-Restricted Retraining Suggests Disease-Specific Signal</title><p>To rule out the possibility that the post hoc subgroup estimate (0.787) reflects features learned from the age-confounded full cohort, we retrained the AST from scratch using only the 129 participants aged 60 to 80 years (84 PD and 45 controls; mean age gap 1.4 years). We acknowledge that fine-tuning all 86.4 million AST parameters on 129 participants carries substantial overfitting risk despite regularization measures (differential learning rate, focal loss, early stopping, and SpecAugment). Transfer learning from AudioSet pretraining mitigates this partially, but the effective number of freely optimized parameters relative to the sample size remains unfavorable. This retrained model achieved an OOF AUC of 0.798 (bootstrap 95% CI 0.710-0.878; mean fold AUC 0.805, SD 0.085), while age-only logistic regression on the same cohort yielded an AUC of 0.555 (near chance). Fold-level AUCs ranged from 0.667 to 0.889 (SD 0.085), indicating substantial instability. This high variance suggests the retrained model&#x2019;s OOF AUC of 0.798 should be interpreted cautiously. The retrained AUC (0.798) closely matches the post hoc estimate (0.787), suggesting that the full-cohort model&#x2019;s age-restricted performance reflects disease-specific features rather than residual age-correlated signal from training. This result directly addresses the concern that post hoc subgroup analysis may overestimate performance due to features learned during confounded training (<xref ref-type="table" rid="table3">Table 3</xref>).</p></sec><sec id="s3-3-4"><title>Country-Level Analysis Reveals Site-Diagnosis Confounding</title><sec id="s3-3-4-1"><title>Overview</title><p>Bridge2AI v3.0.0 does not expose explicit site identifiers, but the country of origin (United States and Canada) provides a coarse proxy for institutional variation. Among the 253 PD cohort participants with demographic data, 191 are from the United States (44 PD and 147 controls) and 62 from Canada (62 PD and 0 controls). The complete absence of Canadian controls means that country perfectly predicts diagnosis for Canadian participants, introducing site-level confounding that compounds the age effect. Restricting to US-only participants, the AST achieved an OOF AUC of 0.779 (age gap 25.0 years); under age restriction (ages 60&#x2010;80 years, n=84: 39 PD, 45 controls; age gap 0.95 years), the US-only AUC was 0.726. The lower US-only age-restricted AUC (0.726) compared to the full-cohort age-restricted AUC (0.787) likely reflects the removal of the perfectly confounded Canadian subsample. These findings demonstrate that site-level demographic imbalance, here, an entire country contributing only cases, can further inflate screening performance beyond the age confound alone. All 62 Canadian participants in the PD cohort are PD cases with zero controls; similarly, all 70 Canadian dementia participants are cases. Consequently, models may partially learn country-of-origin rather than disease-specific features.</p></sec><sec id="s3-3-4-2"><title>Propensity-Matched Analysis (PD)</title><p>Within the US-only ages 60&#x2010;80 years PD subgroup (n=84; 39 PD and 45 controls), 1:1 nearest-neighbor propensity-score matching on age and sex (logistic-regression propensity score; caliper 0.2&#x00D7;SD; no replacement) produced 28 matched pairs (72% of cases matched within caliper, 28/39); the postmatch standardized mean differences were 0.095 for age and 0.071 for sex, both below the conventional 0.1 balance threshold. For the matched cohort, predictions were the existing OOF probabilities from the full-cohort AST; no model was retrained on the matched subset. Matching therefore balances the evaluation distribution on age and sex but does not remove demographic structure that the model may have learned during full-cohort training. On this matched cohort (n=56), the AST achieved an AUC of 0.760 (95% CI 0.628-0.877), while age-only and metadata classifiers were at chance (0.504 and 0.499). DeLong test for AST versus age-only on the matched cohort yielded &#x0394; AUC +0.256 (95% CI +0.038 to +0.475; <italic>P</italic>=.02; q=0.039). This converges with the post hoc age-restricted estimate (0.787) and the de novo retrained estimate (0.798). We note that the matched (0.760) and post hoc (0.787) estimates both reuse the full-cohort model&#x2019;s OOF predictions and are therefore not mutually independent; the de novo retrained estimate (0.798), which never observed the confounded full cohort during training, is the independent corroboration. Taken together, the matched result shows that the AST&#x2019;s discrimination persists on a cohort balanced on age and sex.</p></sec><sec id="s3-3-4-3"><title>Dementia Propensity Matching Infeasible</title><p>The US-only ages 60&#x2010;80 years dementia subgroup contained only 2 cases (vs 45 controls), precluding propensity matching. The full age-restricted dementia AUC of 0.809 therefore necessarily includes Canadian participants, where 100% are cases. This means that our estimate is at least partially confounded by site-level case-control imbalance. We caveat the dementia disease-specific interpretation accordingly.</p></sec></sec></sec><sec id="s3-4"><title>Dementia</title><p>The dementia cohort exhibits an even larger age gap (30.2 years; cases: mean 74.9, SD 7.6 vs controls: mean 44.7, SD 19.3). Age-only logistic regression achieved an AUC of 0.905 for dementia classification, exceeding the AST&#x2019;s 0.895, mirroring the PD pattern, although DeLong test indicates the difference is not statistically significant (&#x0394;AUC=&#x2013;0.009, 95% CI &#x2013;0.067 to +0.049; <italic>P</italic>=.75; q=0.826; <xref ref-type="table" rid="table4">Table 4</xref>). Under age restriction (60&#x2010;80 years; 95 participants: 50 dementia, 45 controls; mean age gap 2.5 years), the AST retained an AUC of 0.809 while age-only classification dropped to 0.598. We note that all 70 Canadian dementia participants are cases with zero controls, identical in confounding structure to the PD cohort. The dementia age-restricted AUC of 0.809 is therefore particularly susceptible to site confounding and should be interpreted with corresponding caution.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Demographic confounding analysis for dementia screening (Bridge2AI<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> v3.0.0; 220 participants; 5-fold stratified cross-validation, participant-level, out-of-fold)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Cohort</td><td align="left" valign="bottom">N</td><td align="left" valign="bottom">AUC-ROC<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Full cohort (age gap=30.2 years<italic>)</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age only (raw)</td><td align="left" valign="top">All</td><td align="left" valign="top">220</td><td align="left" valign="top">0.905</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Metadata only (age+sex)</td><td align="left" valign="top">All</td><td align="left" valign="top">220</td><td align="left" valign="top">0.901</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AST<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup> (spectrogram)</td><td align="left" valign="top">All</td><td align="left" valign="top">220</td><td align="left" valign="top">0.895</td></tr><tr><td align="left" valign="top" colspan="4">Age-restricted subgroup (age gap=6.3 years<italic>)</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AST (spectrogram)</td><td align="left" valign="top">Ages 55&#x2010;85</td><td align="left" valign="top">121</td><td align="left" valign="top">0.839</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age only (raw)</td><td align="left" valign="top">Ages 55&#x2010;85</td><td align="left" valign="top">121</td><td align="left" valign="top">0.727</td></tr><tr><td align="left" valign="top" colspan="4">Strictly age-restricted subgroup (age gap=2.5 years<italic>)</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AST (spectrogram)</td><td align="left" valign="top">Ages 60&#x2010;80</td><td align="left" valign="top">95</td><td align="left" valign="top">0.809</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age only (raw)</td><td align="left" valign="top">Ages 60&#x2010;80</td><td align="left" valign="top">95</td><td align="left" valign="top">0.598</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Metadata only (age+sex)</td><td align="left" valign="top">Ages 60&#x2010;80</td><td align="left" valign="top">95</td><td align="left" valign="top">0.587</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Bridge2AI: Bridge to Artificial Intelligence.</p></fn><fn id="table4fn2"><p><sup>b</sup> Age-only AUC (0.905) exceeds the AST (0.895); under age restriction (60&#x2010;80 years, n=95), age-only drops to 0.598 while the AST retains 0.809.</p></fn><fn id="table4fn3"><p><sup>c</sup>AUC-ROC: area under the receiver operating characteristic curve. </p></fn><fn id="table4fn4"><p><sup>d</sup>AST: audio spectrogram transformer.</p></fn></table-wrap-foot></table-wrap><p>The consistency of these findings across 2 conditions sharing the same control group, both showing age-only AUC exceeding the AST and both showing complete Canadian site confounding (Canada contributes only cases), strongly suggests that demographic confounding is a systematic property of this dataset&#x2019;s design rather than a condition-specific artifact. We emphasize that, unlike PD, the dementia age-restricted AUC (0.809) cannot be attributed to a disease-specific signal once site confounding is considered, because the US-only dementia subgroup contains too few cases to isolate it.</p></sec><sec id="s3-5"><title>Cross-Version Preprocessing Robustness</title><p>Bridge2AI v3.0.0 is a cumulative release that subsumes v2.0.1: session-level identifier matching confirms that 441 of 442 v2.0.1 participants reappear in v3.0.0 under reanonymized participant identifiers, consistent with the PhysioNet release notes (v3.0.0 adds &#x201C;an additional 391 participants&#x201D; to the 442 in v2.0.1). The two releases therefore represent the same participants recorded with the same voice tasks, but with spectrograms extracted through different pipelines (60 vs 201 Mel bins) and different phenotype annotation structures. To assess the sensitivity of learned representations to preprocessing, the 5-fold models trained on the 253-participant v3.0.0 spectrograms were applied as an ensemble to the v2.0.1 spectrograms of the same participants (442 participants: 61 PD and 381 non-PD).</p><p>v2.0.1 spectrograms were resized to (128 &#x00D7;1024) and normalized using each fold&#x2019;s training statistics. The ensemble achieved a participant-level AUC of 0.618, a substantial drop from both the full-cohort OOF AUC (0.843) and the age-restricted AUC (0.787). This performance drop likely reflects the acoustic mismatch between upsampled (v3.0.0, 60 to 128 bins) and downsampled (v2.0.1, 201 to 128 bins) spectrograms; the model was trained on characteristically smooth interpolation patterns absent from the v2.0.1 data, rather than a pure generalization failure.</p><p>Because v2.0.1 contains the same participants as a subset of v3.0.0, this drop cannot be attributed to population differences. Instead, it isolates the effect of spectrogram extraction: the model&#x2019;s learned features do not transfer across preprocessing pipelines, suggesting sensitivity to low-level spectral representation rather than robust encoding of vocal biomarkers. This brittleness reinforces the broader concern that reported within-dataset performance may reflect artifacts of the specific feature extraction pipeline as much as genuine disease signal.</p></sec><sec id="s3-6"><title>AST Screening Performance</title><sec id="s3-6-1"><title>Overview</title><p>Having established the confounding context, we report the AST screening results (<xref ref-type="table" rid="table5">Table 5</xref>). Per-fold metrics are in Table S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> and Figure S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Aggregated out-of-fold AST<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup> screening performance for PD<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup> and dementia (Bridge2AI<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup> v3.0.0; 5-fold stratified cross-validation, participant-level)<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom">Parkinson disease</td><td align="left" valign="bottom">Dementia</td></tr></thead><tbody><tr><td align="left" valign="top">Participants (cases and controls)</td><td align="left" valign="top">253 (106, 147)</td><td align="left" valign="top">221 (73, 148)</td></tr><tr><td align="left" valign="top">AUC-ROC<sup><xref ref-type="table-fn" rid="table5fn5">e</xref></sup></td><td align="left" valign="top">0.843</td><td align="left" valign="top">0.895</td></tr><tr><td align="left" valign="top">Age-restricted AUC<sup><xref ref-type="table-fn" rid="table5fn6">f</xref></sup> (60-80)</td><td align="left" valign="top">0.787</td><td align="left" valign="top">0.809</td></tr><tr><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">0.686</td><td align="left" valign="top">0.773</td></tr><tr><td align="left" valign="top">Recall</td><td align="left" valign="top">0.679</td><td align="left" valign="top">0.863</td></tr><tr><td align="left" valign="top">Precision</td><td align="left" valign="top">0.692</td><td align="left" valign="top">0.700</td></tr><tr><td align="left" valign="top">LOFO<sup><xref ref-type="table-fn" rid="table5fn7">g</xref></sup> threshold, mean (SD)</td><td align="left" valign="top">0.509 (0.034)</td><td align="left" valign="top">0.459 (0.030)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>AST: audio spectrogram transformer.</p></fn><fn id="table5fn2"><p><sup>b</sup>PD: Parkinson disease.</p></fn><fn id="table5fn3"><p><sup>c</sup>Bridge2AI: Bridge to Artificial Intelligence.</p></fn><fn id="table5fn4"><p><sup>d</sup> Full-cohort areas under the curve are inflated by age confounding (27.5&#x2010;30.2 year gaps); age-restricted areas under the curve (<xref ref-type="table" rid="table4">Table 4</xref>) provide less confounded estimates.</p></fn><fn id="table5fn5"><p><sup>e</sup>AUC-ROC: area under the receiver operating characteristic curve. </p></fn><fn id="table5fn6"><p><sup>f</sup>AUC: area under the curve.</p></fn><fn id="table5fn7"><p><sup>g</sup>LOFO: leave-one-fold-out.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6-2"><title>PD</title><p>Mean per-fold AUC-ROC was 0.866 (SD 0.037; 95% CI 0.820-0.912). Aggregated OOF predictions yielded an AUC-ROC of 0.843; under the LOFO threshold protocol (mean threshold 0.509, SD 0.034), <italic>F</italic><sub>1</sub>-score=0.686, precision=0.692, recall=0.679, specificity=0.782 (Table S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>; <xref ref-type="fig" rid="figure4">Figure 4</xref>). Under the LOFO protocol, each fold's threshold was derived from the OOF predictions (Youden J statistic) of the remaining folds rather than from within-fold training partitions, so no participant's threshold depends on their own prediction. Because threshold selection still uses held-out predictions from other participants, the reported <italic>F</italic><sub>1</sub>-score, precision, and recall remain slightly optimistic; AUC is threshold-independent and unaffected. As shown in the &#x201C;Age Confounding Dominates Screening Performance&#x201D; section, the age-restricted estimate (AUC 0.787) should be considered the more reliable indicator of disease-specific performance.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Participant-level audio spectrogram transformer classification performance for Parkinson disease (Bridge2AI v3.0.0; 253 participants; 5-fold stratified cross-validation). (A) Receiver operating characteristic curve (out-of-fold area under the curve 0.843; age-restricted estimate 0.787; <xref ref-type="table" rid="table3">Table 3</xref>). (B) Precision-recall curve. (C) Confusion matrix at the mean leave-one-fold-out threshold (0.509), selected by the Youden J statistic. (D) Per-fold area under the curve distribution (mean 0.866, SD 0.037). AP: average precision; AUC: area under the curve; AUC-ROC: area under the receiver operating characteristic curve; LOFO: leave-one-fold-out; PD: Parkinson disease; ROC: receiver operating characteristic.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95609_fig04.png"/></fig></sec><sec id="s3-6-3"><title>Dementia</title><p>Mean per-fold AUC-ROC was 0.901 (SD 0.032; 95% CI 0.862-0.940). Aggregated OOF evaluation produced an AUC-ROC of 0.895; under the LOFO threshold protocol (mean threshold 0.459, SD 0.030), <italic>F</italic><sub>1</sub>-score=0.773, precision=0.700, recall=0.863, specificity=0.818. The full-cohort AUC is inflated by a 30.2-year age gap; the age-restricted estimate (AUC 0.809) is more reliable (see the &#x201C;Age Confounding Dominates Screening Performance&#x201D; section).</p></sec></sec><sec id="s3-7"><title>Baseline Comparisons and Ablation</title><p>To contextualize the AST&#x2019;s performance, additional baselines were evaluated on the PD cohort using identical cross-validation splits (Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> and Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). A logistic regression baseline on spectrogram summary statistics achieved an OOF AUC of 0.585, a frozen AST baseline (linear probing; 0.23% of parameters trainable) achieved 0.776, and a ResNet18 convolutional baseline achieved 0.821. Fine-tuning improved over frozen representations by 0.067 AUC. The fine-tuned AST outperformed all speech-based baselines but was outperformed by age alone (0.875), underscoring the dominance of the demographic confounder.</p><p>An ablation removing SpecAugment during training yielded an OOF AUC of 0.846, nearly identical to 0.843 with augmentation (&#x0394;=0.003). This near-identical performance with and without SpecAugment may indicate that the model relies on coarse spectral features potentially including those correlated with age or recording site, rather than fine-grained spectrotemporal patterns. SpecAugment masks localized time-frequency regions; if classification is driven by global spectral statistics (eg, overall energy distribution or fundamental frequency range), such masking would have minimal impact regardless of sample size.</p></sec><sec id="s3-8"><title>Supporting Analyses</title><p>A PPG [<xref ref-type="bibr" rid="ref25">25</xref>] comparison showed that spectrogram-based representations outperformed phoneme-based inputs for both PD (0.843 vs 0.737) and dementia (0.895 vs 0.816); however, the PPG-AST comparison is confounded by the interpolation of 40-dimensional categorical probability vectors to 128 bins, which distorts the distributional semantics of the input. The lower PPG-AST performance may reflect this preprocessing artifact rather than an inherent architectural limitation. The 1D CNN on native 40-dimensional PPGs (AUC 0.786) avoids this issue and provides a fairer phoneme-level baseline (Table S9 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). A sensitivity analysis using all available speech tasks (9331 recordings vs 2131 for selected tasks) yielded an OOF AUC of 0.942, though this elevated performance likely reflects task completion patterns acting as a data leak; the pattern of completed versus missing tasks may indirectly encode participant age and condition (Table S10 and Figure S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Both models showed reasonable calibration (PD Brier score 0.166; dementia 0.130; Table S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>; Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), and all 8 selected tasks exceeded chance-level PD discrimination (Figure S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). <xref ref-type="table" rid="table6">Table 6</xref> summarizes AST screening AUCs for PD across all analytical conditions reported above, from the full cohort (0.843) through age restriction and propensity matching (0.760) to the cross-version preprocessing and all-tasks sensitivity analyses.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Summary of AST<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup> screening AUC<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup> across analytical conditions for Parkinson disease (Bridge2AI<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup> Voice Dataset v3.0.0; 5-fold stratified cross-validation, participant-level, out-of-fold)<sup><xref ref-type="table-fn" rid="table6fn4">d</xref></sup>.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Condition</td><td align="left" valign="bottom">N</td><td align="left" valign="bottom">AST AUC (95% CI)</td><td align="left" valign="bottom">Age-only AUC</td><td align="left" valign="bottom">DeLong P (BH<sup><xref ref-type="table-fn" rid="table6fn5">k</xref></sup> q)</td></tr></thead><tbody><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Full cohort</td><td align="left" valign="top">252</td><td align="left" valign="top">0.843 (0.791&#x2010;0.890)</td><td align="left" valign="top">0.875</td><td align="left" valign="top">.36 (0.462)<sup><xref ref-type="table-fn" rid="table6fn6">f</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age-restricted (55-85)</td><td align="left" valign="top">151</td><td align="left" valign="top">0.805 (&#x2014;<sup><xref ref-type="table-fn" rid="table6fn7">g</xref></sup>)</td><td align="left" valign="top">0.655</td><td align="left" valign="top">.007 (0.016)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age-restricted (60-80)</td><td align="left" valign="top">129</td><td align="left" valign="top">0.787 (0.695&#x2010;0.863)</td><td align="left" valign="top">0.568</td><td align="left" valign="top">.001 (0.005)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age-restricted retrained</td><td align="left" valign="top">129</td><td align="left" valign="top">0.798 (0.710&#x2010;0.878)</td><td align="left" valign="top">0.555</td><td align="left" valign="top">.002<sup><xref ref-type="table-fn" rid="table6fn8">h</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>US-only full</td><td align="left" valign="top">191</td><td align="left" valign="top">0.779 (&#x2014;)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>US-only age-restricted</td><td align="left" valign="top">84</td><td align="left" valign="top">0.726 (&#x2014;)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><italic>Propensity-matched (United States, age + sex)</italic></td><td align="left" valign="top"><italic>56</italic></td><td align="left" valign="top"><italic>0.760 (0.628&#x2010;0.877)</italic></td><td align="left" valign="top"><italic>0.504</italic></td><td align="left" valign="top"><italic>.021 (0.039)</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cross-version preprocessing (v2.0.1)</td><td align="left" valign="top">442</td><td align="left" valign="top">0.618 (&#x2014;)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>All available speech tasks</td><td align="left" valign="top">253</td><td align="left" valign="top">0.942 (&#x2014;)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>AST: audio spectrogram transformer.</p></fn><fn id="table6fn2"><p><sup>b</sup>AUC: area under the curve.</p></fn><fn id="table6fn3"><p><sup>c</sup>Bridge2AI: Bridge to Artificial Intelligence. </p></fn><fn id="table6fn4"><p><sup>d</sup>The italicized row marks the analysis with the strongest formal evidence of disease-specific signal beyond demographics (1:1 propensity matching on age and sex within the US-only ages 60&#x2010;80 Parkinson disease subgroup; AST vs age-only DeLong <italic>P</italic>=.02; q=0.039). AUC 95% CIs are bootstrap intervals (2000 resamples) where computed; q=Benjamini-Hochberg FDR-adjusted P. The ages 55&#x2010;85 row is a wider, overlapping version of the prespecified ages 60&#x2010;80 age restriction. The progression from 0.843 (full cohort) to 0.760 (matched) tracks the removal of demographic confounders. Cross-version preprocessing AUC (0.618) reflects spectrogram-pipeline sensitivity rather than population generalization; all-tasks AUC (0.942) suffers from task-completion missingness (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></fn><fn id="table6fn5"><p><sup>e</sup>BH: Benjamini-Hochberg.</p></fn><fn id="table6fn6"><p><sup>f</sup>Not significant.</p></fn><fn id="table6fn7"><p><sup>g</sup>Not applicable.</p></fn><fn id="table6fn8"><p><sup>h</sup>De novo retrained audio spectrogram transformer versus age-only on the retrained cohort; reported for completeness and not part of the 9-comparison Benjamini-Hochberg family, hence shown uncorrected.</p></fn></table-wrap-foot></table-wrap><p>Attention map analysis of the fold-1 model (Section 7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> and Figure S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) revealed no statistically significant global differences in attention patterns between PD and control groups (permutation test: <italic>P</italic>=.16 for the full cohort; <italic>P</italic>=.88 for the age-restricted subgroup). Exploratory frequency-band analysis identified elevated PD attention in the lowest frequency band (~0&#x2010;600 Hz; Cohen <italic>d</italic>=0.57), but this finding did not survive correction for the 6 bands tested and should be interpreted as hypothesis-generating rather than confirmatory.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><p>The central finding of this study is that age confounding dominates screening performance for both PD and dementia in the Bridge2AI Voice Dataset: a simple age threshold outperforms a fine-tuned 86.4-million-parameter AST for both conditions (age-only AUC 0.875 vs AST 0.843 for PD; 0.905 vs 0.895 for dementia). The consistency of this pattern across 2 conditions sharing the same control group strongly suggests that the confounding is a systematic property of the dataset&#x2019;s demographic structure rather than a condition-specific artifact. This result challenges the interpretation of high classification metrics commonly reported in the voice-based disease screening literature and has broad implications for how studies in this field should be designed, evaluated, and reported.</p></sec><sec id="s4-2"><title>The Scope of Demographic Confounding</title><p>The finding that age alone achieves an AUC of 0.875 for PD classification, exceeding the AST&#x2019;s 0.843, is striking but not surprising in retrospect. PD onset typically occurs at approximately 60 years of age [<xref ref-type="bibr" rid="ref13">13</xref>], while volunteer control cohorts in voice research typically skew younger. The resulting age gap creates a statistical shortcut: any classifier that captures age-correlated features (including age-related voice changes such as decreased pitch, increased jitter, and reduced harmonic-to-noise ratio [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]) will achieve high discrimination without learning anything about PD itself. The dementia cohort confirms this is not condition-specific: with a 30.2-year age gap, age alone achieves 0.905, again exceeding the AST (0.895). Under age restriction (60 to 80 years), the dementia AST AUC was 0.809 while age-only dropped to 0.598; however, unlike PD, this age-restricted estimate cannot be interpreted as disease-specific, because all 70 Canadian dementia participants are cases and only 2 of 47 US-only ages 60&#x2010;80 years dementia participants are cases. The 0.809 figure therefore conflates disease signal with site and effectively functions as a country classifier, and no site-controlled dementia estimate is derivable from this dataset. Statistically, age-only and AST classifiers were not distinguishable in the full cohort for either condition (PD: <italic>P</italic>=.36, q=0.462; dementia: <italic>P</italic>=.75, q=0.826), meaning the AST contributes no measurable discrimination beyond demographic features at full-cohort level. Under age-restricted evaluation (60&#x2010;80 years), the AST exceeded age-only by a wide and statistically significant margin (PD: &#x0394; AUC +0.225, 95% CI +0.089 to +0.361, <italic>P</italic>=.001, q=0.005; dementia: +0.215, 95% CI +0.056 to +0.375, <italic>P</italic>=.008, q=0.018), confirming that the model captures genuine disease-specific information that emerges only after the age confound is removed. We note that the PD and dementia cohorts share the same 148-person control group (mean age 44.7, SD 19.3 years). Consequently, the parallel confounding pattern across both conditions is a single structural observation, not 2 independent validations. The converging evidence is informative; it shows age predicts both conditions&#x2019; labels, but the shared controls mean both age-only baselines are mechanistically coupled.</p><p>This problem is not unique to the Bridge2AI dataset. Brenner et al [<xref ref-type="bibr" rid="ref18">18</xref>] demonstrated that age-based selection bias inflates PD voice classification on the mPower dataset, and Hire&#x0161; et al [<xref ref-type="bibr" rid="ref27">27</xref>] showed that cross-dataset voice-based PD detection yields substantially degraded performance even when within-dataset accuracy is high, consistent with our cross-version finding (AUC 0.618). We have found &#x201C;shortcuts&#x201D; in imaging [<xref ref-type="bibr" rid="ref28">28</xref>], and others have found demographic shortcuts that AI models often use [<xref ref-type="bibr" rid="ref29">29</xref>]. Many PD voice studies use datasets with similar or unreported demographic structures. Rahman et al [<xref ref-type="bibr" rid="ref21">21</xref>] achieved AUC 0.753 on a web-based speech task (726 participants; PD: mean age 65.9 years vs control: mean age 58.0 years) and acknowledged age and sex as confounders, but their mitigation was limited to excluding participants younger than 50 years; they did not test whether age alone could predict PD status. Of 13 representative studies surveyed (Table S12 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref36">36</xref>]), only 7 report age distributions for both cases and controls, none performed a deliberate age-restricted subgroup evaluation, and none tested a demographic-only baseline. Studies reporting 90% to 97% classification accuracy [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref34">34</xref>] typically do not assess the contribution of age to their results (Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]). Without such controls, it is impossible to determine what fraction of reported performance reflects genuine disease-specific signal versus demographic confounding. We note that the TRIPOD+AI guidelines [<xref ref-type="bibr" rid="ref16">16</xref>] explicitly recommend assessment of demographic factors in prediction models, yet this analysis is largely absent from the voice biomarker literature.</p><p>From a clinical perspective, these findings urge caution in interpreting reported screening performance for voice-based PD and dementia tools. An AUC of 0.787 under age-restricted conditions is substantially lower than the 0.843 reported on the full cohort; yet, it reflects genuine disease-specific discriminative ability rather than demographic artifact.</p><p>The age-restricted subgroup analysis provides a practical solution. By restricting evaluation to participants aged 60 to 80 (PD: mean age 71.4, SD 5.8 vs control: mean age 70.0, SD 6.5) years, we effectively eliminate age as a confounder (age-only AUC drops to 0.568, near chance). Under these conditions, the AST retains an AUC of 0.787, suggesting that the model does capture genuine disease-specific acoustic features, but at a substantially lower level than the full-cohort metric suggests. Retraining the AST from scratch on the age-restricted subgroup corroborates this estimate (AUC 0.798), ruling out the concern that the post hoc result reflects features learned during confounded training. This convergence of post hoc evaluation (0.787) and de novo retraining (0.798) suggests that voice-based features may contain disease-relevant information at approximately this performance level, though confirmation requires prospective validation on independent, age-balanced cohorts.</p><p>The confounding problem extends beyond age to site-level demographic imbalance. All 62 Canadian participants in the PD cohort are cases (zero controls), creating perfect site-diagnosis confounding. Restricting to US-only participants under age restriction yields an AUC of 0.726, lower than the full-cohort age-restricted AUC of 0.787, suggesting that the perfectly confounded Canadian subsample inflates even the age-restricted metric. The complete case-control site confounding (Canada contributes only cases for both conditions) means the age-restricted AUC of 0.787 may still be partially inflated by site-specific acoustic properties. The US-only age-restricted estimate of 0.726, while lower, is arguably the least confounded performance estimate available from this dataset. We recommend this figure be given greater interpretive weight. This finding highlights that multisite datasets may introduce additional confounders that are not resolved by age restriction alone.</p><p>Cross-version preprocessing robustness testing reinforces the fragility of within-dataset metrics: performance dropped from 0.843 to 0.618 when v3.0.0-trained models were applied to v2.0.1 spectrograms of the same participants (442 individuals with spectrograms reextracted at 201 vs 60 Mel bins). Because v2.0.1 participants are a subset of v3.0.0 (confirmed via session identifier matching), this drop isolates the effect of the spectrogram extraction pipeline rather than population differences, revealing that the model&#x2019;s learned representations are brittle to preprocessing changes, a concern for clinical deployment where feature extraction standardization cannot be guaranteed.</p></sec><sec id="s4-3"><title>Implications for the Field</title><p>The demographic confounding problem identified here likely extends beyond PD. Any condition that predominantly affects older or younger adults (eg, Alzheimer disease, amyotrophic lateral sclerosis, and attention-deficit disorders) will produce age-imbalanced datasets when cases are compared against convenience-sampled controls. Voice changes with age are well documented [<xref ref-type="bibr" rid="ref15">15</xref>], creating a persistent confounder that standard evaluation metrics cannot resolve. Concurrent work achieving 0.89 AUC-ROC on Bridge2AI [<xref ref-type="bibr" rid="ref36">36</xref>] has not been evaluated under age restriction, and self-supervised models [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>] face the same concern.</p><p>We propose that future voice biomarker studies adopt three minimum reporting requirements: (1) report the age distribution of cases and controls, (2) train and evaluate a demographic-only baseline using the same cross-validation splits, and (3) report age-restricted subgroup performance alongside full-cohort metrics. These additions require minimal computation but would substantially improve interpretability. We further recommend that dataset creators prioritize age-restricted cohort designs at the recruitment stage.</p><p>The educational origin of these findings is instructive: the age confounding was first identified through bias auditing exercises designed for early-career researchers at the Emory Health AI Summer School &#x0026; Datathon [<xref ref-type="bibr" rid="ref22">22</xref>], suggesting that structured training in demographic bias assessment, rather than specialized methodological expertise, is sufficient to uncover these issues. This supports the integration of bias auditing protocols into standard machine learning training curricula for biomedical researchers.</p></sec><sec id="s4-4"><title>Limitations</title><p>The principal evidence for a residual disease-specific signal in this study comes from 3 subgroup analyses with substantially reduced sample sizes, and several sources of confounding could not be fully eliminated within the constraints of the available data. We organize the limitations below according to the analytical layer they affect.</p></sec><sec id="s4-5"><title>Sample Size and Statistical Interpretation</title><p>The principal evidence for a disease-specific PD signal beyond demographics rests on 3 convergent subgroup analyses: post hoc age-restricted evaluation (129 participants, AUC 0.787), de novo retraining on the age-restricted subgroup (the same 129, AUC 0.798), and 1:1 propensity-score matching on age and sex within the US-only ages 60 to 80 PD subgroup (28 matched pairs, AUC 0.760). The triangulation across these 3 procedures is the strongest argument that the AST captures information beyond demographics; we note, however, that only the de novo retraining is fully independent of the confounded full-cohort model, whereas the post hoc and propensity-matched estimates share the full-cohort OOF predictions, but each individual estimate has wide CIs (eg, bootstrap 95% CI 0.695-0.863 for the post hoc estimate) and the retrained model exhibited substantial fold-to-fold variance (SD 0.085) reflecting the unfavorable ratio of 86.4 million parameters to 129 participants. The corresponding statistical tests must also be interpreted in this context: the full-cohort DeLong tests were not statistically significant (PD: <italic>P</italic>=.36, q=0.462; dementia: <italic>P</italic>=.75, q=0.826), which is a failure to reject equivalence rather than a positive demonstration of it; the age-restricted superiority of the AST over age alone (PD: <italic>P</italic>=.001, q=0.005; dementia: <italic>P=</italic>.008, q=0.018) and the propensity-matched DeLong <italic>P</italic>=.02, q=0.039 are statistically significant but obtained on small samples and remain vulnerable to sample-specific effects. Replication on larger, prospectively recruited age-balanced cohorts is necessary, and future work should examine whether progressive layer freezing or adapter-based fine-tuning yields more stable estimates in the small-sample regime. A further source of optimism is that model checkpoints were selected by early stopping on each fold&#x2019;s held-out partition, which was then aggregated to report OOF performance; this couples model selection to the evaluation set and biases the reported AUCs upward. We did not implement nested cross-validation owing to the computational cost of repeatedly fine-tuning an 86.4-million-parameter model. Critically, this bias inflates the AST&#x2019;s metrics, not the demographic-only baselines (which use independent logistic-regression cross-validation); it therefore cannot explain why age alone matches or exceeds the AST on the full cohort, and the true disease-specific signal is, if anything, slightly lower than the age-restricted estimates reported here.</p></sec><sec id="s4-6"><title>Residual Site Confounding and Dementia Estimate</title><p>Although the US-only ages 60 to 80 years estimate (AUC 0.726) substantially reduces the site confound introduced by the all-case Canadian subsample, it does not eliminate it. Bridge2AI v3.0.0 does not expose explicit site identifiers, so we used country of origin as a coarse proxy; institutional, device, or recording-environment variation among the 4 US sites remains unmeasured and could still partially inflate the residual signal. We therefore present 0.726 as the most conservative estimate available from this dataset, not as an unconfounded one. The corresponding dementia analysis is more constrained still: only 2 of the 47 US-only ages 60 to 80 dementia participants were cases, which precluded propensity matching and means the dementia age-restricted AUC of 0.809 necessarily includes the entirely-case Canadian subsample. A properly site-adjusted dementia disease-specific estimate analogous to the PD estimate of 0.726 cannot be derived from this dataset, and demonstrating an unconfounded dementia voice signal will require a cohort with substantially more US-based dementia cases in the 60 to 80 age range.</p></sec><sec id="s4-7"><title>Demographic and Clinical Covariates Not Incorporated</title><p>Bridge2AI records participant race, ethnicity, primary language, education level, Hoehn-Yahr stage (PD), Montreal Cognitive Assessment scores, and medication state, none of which were incorporated into the present analysis. Voice biomarkers are known to encode race, accent, and language background, and the AST backbone was pretrained on AudioSet, which may carry its own demographic biases independent of any task-specific signal. Diagnoses were treated as binary, which means residual disease-specific performance may disproportionately reflect more severely affected cases and will not generalize uniformly across the disease severity spectrum. Medication effects, particularly levodopa for PD, were not controlled, and the dataset comprises only English-language speakers, limiting cross-linguistic generalizability. Severity-stratified, demographic-stratified, and medication status&#x2013;stratified analyses are essential prerequisites for any clinical translation claim.</p></sec><sec id="s4-8"><title>Generalizability Beyond This Dataset and Architecture</title><p>The parallel confounding pattern observed across PD and dementia (which share the same control group) suggests a structural property of the Bridge2AI dataset rather than a condition-specific artifact, but the demographic-shortcut phenomenon was not quantified for other voice datasets in this study. Our survey of 13 published voice-PD studies (Table S12 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) found that fewer than half report case and control age distributions and none performed a demographic-only baseline, so the broader scope of confounding in the field remains unmeasured. The confounding analysis also used a single fine-tuned AST backbone; whether self-supervised speech foundation models, multitask learners, or alternative spectrotemporal architectures exhibit the same age-shortcut sensitivity is an open question. The PPG-AST comparison reported here is itself confounded by the interpolation of categorical phoneme posteriors and so cannot adjudicate architectural suitability. A further generalizability concern is task aggregation: participant-level predictions were computed as unweighted means across the 8 selected tasks, with recording counts ranging from 247 to 290 per task, so participants who completed more tasks contributed to a different mixture of predictions than those who completed fewer. If task completion correlates with age, motor function, or cognitive status, some of the residual disease-specific signal could reflect missingness patterns rather than acoustic content, as the elevated all-tasks AUC (0.942, Table S10 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) most parsimoniously suggests. Task-weighted aggregation and an age-restricted evaluation of the all-tasks pipeline are important next steps. The minimum-duration filter (recordings with fewer than 100 timeframes excluded) operated at the recording level and removed only 8 of 2139 selected-task recordings (0.4%); no participant was excluded entirely (253 of 253 retained). The excluded recordings were modestly skewed toward PD cases (5 of 8) with higher mean age (69.4 vs 53.7 years for the 3 excluded control recordings), consistent in direction with duration-related survivorship but negligible in magnitude. This exclusion is therefore a minor component of the broader task-missingness mechanism discussed above rather than an independent source of bias (Table S11 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s4-9"><title>Cross-Version</title><p>We further note that the v2.0.1 spectrograms were standardized using each fold&#x2019;s v3.0.0 training statistics. Because the two versions differ in base Mel resolution (60 bins upsampled to 128 vs 201 bins downsampled to 128), their raw intensity distributions differ, so applying v3.0.0 normalization parameters to v2.0.1 inputs may itself amplify the domain shift. The reported AUC of 0.618 should therefore be read as a combined index of representational brittleness and normalization mismatch rather than population non-generalization alone.</p></sec><sec id="s4-10"><title>Future Directions</title><p>Having demonstrated that age-restricted retraining confirms disease-specific signal (AUC 0.798 vs post hoc 0.787), the most impactful follow-up would be prospectively designed age-restricted cohort recruitment, which would address the confounder at the data collection stage rather than requiring post hoc correction or subgroup analysis on underpowered samples. Our country-level analysis revealed complete site-diagnosis confounding for Canadian participants; access to explicit site identifiers within Bridge2AI (which spans 5 clinical sites) would enable more granular site-level confounding analysis and site-aware model development. A systematic reevaluation of published results in the field, applying the demographic baseline and age-restricted protocol to existing datasets and models, would quantify the scope of the confounding problem across the voice biomarker literature.</p></sec><sec id="s4-11"><title>Conclusions</title><p>This methodological audit of the Bridge2AI Voice Dataset v3.0.0 yields 3 findings that map directly onto the study&#x2019;s stated aims. First, full-cohort screening performance on this dataset is dominated by demographic confounding: age alone is statistically indistinguishable from a fine-tuned 86.4-million-parameter AST for both PD (DeLong: <italic>P</italic>=.36; q=0.462) and dementia (<italic>P</italic>=.75; q=0.826), and a single national subsample (Canadian participants) consists entirely of cases for both conditions. Because the PD and dementia cohorts share the same 148-person control group, the parallel confounding pattern reflects a single dataset-level structural artifact rather than 2 independent confirmations.</p><p>Second, after age restriction, site control, and propensity-score matching on age and sex, residual PD performance ranges from AUC 0.726 to 0.798 and exceeds age-only baselines at the propensity-matched level (<italic>P</italic>=.02; q=0.039), suggesting a residual disease-specific acoustic signal at substantially lower performance than the 0.843 full-cohort metric implies. An analogous unconfounded dementia estimate could not be derived because only 2 of 47 US-only ages 60 to 80 years dementia participants were cases. Cross-version preprocessing testing further showed that the learned representations are brittle to changes in the spectrogram extraction pipeline (AUC 0.618), indicating that within-dataset performance partially reflects features of the specific feature-extraction pipeline rather than robust acoustic biomarkers.</p><p>Third, drawing on these findings and on a survey of 13 published voice-PD studies in which 7 reported case and control age distributions and none performed a demographic-only baseline, we propose 3 minimum reporting standards for voice biomarker research: case and control demographic distributions, demographic-only baselines under identical cross-validation splits, and age-restricted subgroup performance reported alongside full-cohort metrics. Independent prospective validation on age-balanced cohorts, with explicit site identifiers and harmonized acoustic protocols, will be required before voice-based PD or dementia screening can be considered for clinical translation.</p></sec></sec></body><back><ack><p>The authors thank the Bridge2AI-Voice Consortium and all study participants who contributed voice recordings to the dataset. We are particularly grateful to the Bridge2AI Voice Consortium for clarifying the relationship between dataset versions during preparation of this manuscript, which informed our reframing of the v3.0.0-v2.0.1 comparison as a preprocessing-robustness rather than external-validation experiment. Claude Opus 4.7 was used to find bugs in the statistical analysis code, but the original code was written by the authors. ChatGPT 5.5 was used to suggest language improvements in the manuscript, which was then reworked by the authors.</p></ack><notes><sec><title>Funding</title><p>SS, SP, and JWG were supported by grant 1R25OD039834-01 from the National Institutes of Health (NIH) Office of Data Science Strategy (ODSS). JWG receives funding from the National Heart, Lung and Blood Institute (NHLBI) grants R01HL167811 and R01HL177003, and NIH grant 1OT20D038065-01. The Bridge2AI Voice Dataset was generated with support from NIH project number 3OT2OD032720-01S1. The content is solely the responsibility of the authors and does not necessarily represent the official views of the NIH or the Robert Wood Johnson Foundation.</p></sec><sec><title>Data Availability</title><p>The datasets analyzed during this study are available in the PhysioNet repository [<xref ref-type="bibr" rid="ref24">24</xref>]. The Bridge2AI Voice Dataset v3.0.0 (primary analysis) and v2.0.1 (preprocessing robustness analysis) are available under registered access; access requires credentialing through PhysioNet and execution of the Bridge2AI Voice Registered Access Data Use Agreement. No new data were generated by the authors during this study.</p><p>All code for data preprocessing, model training, cross-validation, and evaluation is publicly available at GitHub [<xref ref-type="bibr" rid="ref26">26</xref>]. The repository includes scripts to reproduce the primary and sensitivity analyses reported in this study. No formal study protocol was prepared prior to the analysis, and this study was not preregistered in a prediction model registry. The TRIPOD+AI checklist [<xref ref-type="bibr" rid="ref16">16</xref>] is provided as <xref ref-type="supplementary-material" rid="app2">Checklist 1</xref>.</p></sec></notes><fn-group><fn fn-type="con"><p>Data curation: SS</p><p>Formal analysis: SS</p><p>Investigation: SS, PN</p><p>Methodology: SS, SP</p><p>Software: SS, PN</p><p>Validation: PN</p><p>Visualization: SS</p><p>Funding acquisition: JWG, SP</p><p>Project administration: JWG</p><p>Resources: JWG, SP</p><p>Supervision: SP</p><p>Conceptualization: SP</p><p>Writing &#x2013; original draft: SS</p><p>Writing &#x2013; review &#x0026; editing: PN, JWG, SP</p></fn><fn fn-type="conflict"><p>PN discloses that this study was conducted independently and does not draw upon her work at ConcertAI, where she is employed. Her participation in this study was in a personal capacity and was not funded. JWG serves on several advisory boards, including the American Heart Association (AHA) debiasing clinical care algorithms (DECCA), the American College of Radiology AI advisory council, and the Council of Medical Specialty Societies Equity Initiative. She is a board member of SIIM and an associate editor for RSNA:AI journal. SS and SP declare no competing interests.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">1D</term><def><p>1-dimensional</p></def></def-item><def-item><term id="abb2">AST</term><def><p>audio spectrogram transformer</p></def></def-item><def-item><term id="abb3">AUC</term><def><p>area under the curve</p></def></def-item><def-item><term id="abb4">AUC-ROC</term><def><p>area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb5">Bridge2AI</term><def><p>Bridge to Artificial Intelligence</p></def></def-item><def-item><term id="abb6">CIMDAR-HIVE</term><def><p>Developing a Hive Learning and Datathon Supported Course on Imaging and Multimodal Data for Resource-Limited Institutions</p></def></def-item><def-item><term id="abb7">CNN</term><def><p>convolutional neural network</p></def></def-item><def-item><term id="abb8">FDR</term><def><p>false discovery rate</p></def></def-item><def-item><term id="abb9">GELU</term><def><p>Gaussian Error Linear Unit</p></def></def-item><def-item><term id="abb10">LOFO</term><def><p>leave-one-fold-out</p></def></def-item><def-item><term id="abb11">NIH</term><def><p>National Institutes of Health</p></def></def-item><def-item><term id="abb12">OOF</term><def><p>out-of-fold</p></def></def-item><def-item><term id="abb13">PD</term><def><p>Parkinson disease</p></def></def-item><def-item><term id="abb14">PPG</term><def><p>phonetic posteriorgram</p></def></def-item><def-item><term id="abb15">TRIPOD+AI</term><def><p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis&#x2013;Artificial Intelligence</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Muddaloor</surname><given-names>P</given-names> </name><name name-style="western"><surname>Baraskar</surname><given-names>B</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>H</given-names> </name><etal/></person-group><article-title>The human voice as a digital health solution leveraging artificial intelligence</article-title><source>Sensors (Basel)</source><year>2025</year><month>05</month><day>29</day><volume>25</volume><issue>11</issue><fpage>3424</fpage><pub-id pub-id-type="doi">10.3390/s25113424</pub-id><pub-id pub-id-type="medline">40968958</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xavier</surname><given-names>D</given-names> </name><name name-style="western"><surname>Felizardo</surname><given-names>V</given-names> </name><name name-style="western"><surname>Ferreira</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Voice analysis in Parkinson&#x2019;s disease - a systematic literature review</article-title><source>Artif Intell Med</source><year>2025</year><month>05</month><volume>163</volume><fpage>103109</fpage><pub-id pub-id-type="doi">10.1016/j.artmed.2025.103109</pub-id><pub-id pub-id-type="medline">40132400</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sedigh Malekroodi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>BI</given-names> </name><name name-style="western"><surname>Yi</surname><given-names>M</given-names> </name></person-group><article-title>Voice-based detection of Parkinson&#x2019;s disease using machine and deep learning approaches: a systematic review</article-title><source>Bioengineering (Basel)</source><year>2025</year><month>11</month><day>20</day><volume>12</volume><issue>11</issue><fpage>1279</fpage><pub-id pub-id-type="doi">10.3390/bioengineering12111279</pub-id><pub-id pub-id-type="medline">41301235</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Behroozmand</surname><given-names>R</given-names> </name><name name-style="western"><surname>Shebek</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hansen</surname><given-names>DR</given-names> </name><etal/></person-group><article-title>Sensory-motor networks involved in speech production and motor control: an fMRI study</article-title><source>Neuroimage</source><year>2015</year><month>04</month><day>1</day><volume>109</volume><fpage>418</fpage><lpage>428</lpage><pub-id pub-id-type="doi">10.1016/j.neuroimage.2015.01.040</pub-id><pub-id pub-id-type="medline">25623499</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Van Lancker Sidtis</surname><given-names>D</given-names> </name><name name-style="western"><surname>Rogers</surname><given-names>T</given-names> </name><name name-style="western"><surname>Godier</surname><given-names>V</given-names> </name><name name-style="western"><surname>Tagliati</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sidtis</surname><given-names>JJ</given-names> </name></person-group><article-title>Voice and fluency changes as a function of speech task and deep brain stimulation</article-title><source>J Speech Lang Hear Res</source><year>2010</year><month>10</month><volume>53</volume><issue>5</issue><fpage>1167</fpage><lpage>1177</lpage><pub-id pub-id-type="doi">10.1044/1092-4388(2010/09-0154)</pub-id><pub-id pub-id-type="medline">20643796</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>HT</given-names> </name><name name-style="western"><surname>Kulzhabayeva</surname><given-names>D</given-names> </name><name name-style="western"><surname>Erceg</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Natural speech analysis can reveal individual differences in executive function across the adult lifespan</article-title><source>J Speech Lang Hear Res</source><year>2025</year><month>12</month><day>10</day><volume>68</volume><issue>12</issue><fpage>5708</fpage><lpage>5726</lpage><pub-id pub-id-type="doi">10.1044/2025_JSLHR-24-00268</pub-id><pub-id pub-id-type="medline">41202270</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Gong</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>YA</given-names> </name><name name-style="western"><surname>Glass</surname><given-names>J</given-names> </name></person-group><article-title>AST: audio spectrogram transformer</article-title><conf-name>Interspeech 2021</conf-name><conf-date>Aug 30 to Sep 3, 2021</conf-date><conf-loc>Brno, Czechia</conf-loc><fpage>571</fpage><lpage>575</lpage><pub-id pub-id-type="doi">10.21437/Interspeech.2021-698</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Madusanka</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>BI</given-names> </name></person-group><article-title>Vocal biomarkers for Parkinson&#x2019;s disease classification using audio spectrogram transformers</article-title><source>J Voice</source><year>2024</year><month>12</month><day>10</day><fpage>00388</fpage><lpage>6</lpage><pub-id pub-id-type="doi">10.1016/j.jvoice.2024.11.008</pub-id><pub-id pub-id-type="medline">39665946</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sedigh Malekroodi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Madusanka</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>BI</given-names> </name><name name-style="western"><surname>Yi</surname><given-names>M</given-names> </name></person-group><article-title>Speech-based Parkinson&#x2019;s detection using pre-trained self-supervised automatic speech recognition (ASR) models and supervised contrastive learning</article-title><source>Bioengineering (Basel)</source><year>2025</year><month>07</month><day>1</day><volume>12</volume><issue>7</issue><fpage>728</fpage><pub-id pub-id-type="doi">10.3390/bioengineering12070728</pub-id><pub-id pub-id-type="medline">40722419</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>La Quatra</surname><given-names>M</given-names> </name><name name-style="western"><surname>Turco</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Svendsen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Salvi</surname><given-names>G</given-names> </name><name name-style="western"><surname>Orozco-Arroyave</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Siniscalchi</surname><given-names>SM</given-names> </name></person-group><article-title>Exploiting foundation models and speech enhancement for parkinson&#x2019;s disease detection from speech in real-world operative conditions</article-title><conf-name>Interspeech 2024</conf-name><conf-date>Sep 1-5, 2024</conf-date><conf-loc>Kos Island, Greece</conf-loc><fpage>1405</fpage><lpage>1409</lpage><pub-id pub-id-type="doi">10.21437/Interspeech.2024-522</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>W</given-names> </name><name name-style="western"><surname>Lv</surname><given-names>R</given-names> </name><name name-style="western"><surname>Du</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Parkinson&#x2019;s disease detection using spectrogram-based multi-model feature fusion networks</article-title><source>Front Neurol</source><year>2025</year><volume>16</volume><fpage>1706317</fpage><pub-id pub-id-type="doi">10.3389/fneur.2025.1706317</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Islam</surname><given-names>R</given-names> </name><name name-style="western"><surname>Abdel-Raheem</surname><given-names>E</given-names> </name><name name-style="western"><surname>Tarique</surname><given-names>M</given-names> </name></person-group><article-title>Voice pathology detection using convolutional neural networks with electroglottographic (EGG) and speech signals</article-title><source>Computer Methods and Programs in Biomedicine Update</source><year>2022</year><volume>2</volume><fpage>100074</fpage><pub-id pub-id-type="doi">10.1016/j.cmpbup.2022.100074</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cao</surname><given-names>F</given-names> </name><name name-style="western"><surname>Vogel</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Gharahkhani</surname><given-names>P</given-names> </name><name name-style="western"><surname>Renteria</surname><given-names>ME</given-names> </name></person-group><article-title>Speech and language biomarkers for Parkinson&#x2019;s disease prediction, early diagnosis and progression</article-title><source>NPJ Parkinsons Dis</source><year>2025</year><month>03</month><day>24</day><volume>11</volume><issue>1</issue><fpage>57</fpage><pub-id pub-id-type="doi">10.1038/s41531-025-00913-4</pub-id><pub-id pub-id-type="medline">40128529</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Azadi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Akbarzadeh-T</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Shoeibi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kobravi</surname><given-names>HR</given-names> </name></person-group><article-title>Evaluating the effect of Parkinson&#x2019;s disease on Jitter and Shimmer speech features</article-title><source>Adv Biomed Res</source><year>2021</year><volume>10</volume><fpage>54</fpage><pub-id pub-id-type="doi">10.4103/abr.abr_254_21</pub-id><pub-id pub-id-type="medline">35127581</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xiu</surname><given-names>N</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>L</given-names> </name><etal/></person-group><article-title>A study on voice measures in patients with Parkinson&#x2019;s disease</article-title><source>J Voice</source><year>2024</year><month>06</month><day>17</day><fpage>00168</fpage><lpage>1</lpage><pub-id pub-id-type="doi">10.1016/j.jvoice.2024.05.018</pub-id><pub-id pub-id-type="medline">38890016</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Collins</surname><given-names>GS</given-names> </name><name name-style="western"><surname>Moons</surname><given-names>KGM</given-names> </name><name name-style="western"><surname>Dhiman</surname><given-names>P</given-names> </name><etal/></person-group><article-title>TRIPOD+AI statement: updated guidance for reporting clinical prediction models that use regression or machine learning methods</article-title><source>BMJ</source><year>2024</year><month>04</month><day>16</day><volume>385</volume><fpage>e078378</fpage><pub-id pub-id-type="doi">10.1136/bmj-2023-078378</pub-id><pub-id pub-id-type="medline">38626948</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Russell</surname><given-names>L</given-names> </name><name name-style="western"><surname>Bensoussan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Developing age-specific protocols for pediatric voice databases in artificial intelligence research</article-title><source>Int J Pediatr Otorhinolaryngol</source><year>2025</year><month>09</month><volume>196</volume><fpage>112455</fpage><pub-id pub-id-type="doi">10.1016/j.ijporl.2025.112455</pub-id><pub-id pub-id-type="medline">40680403</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brenner</surname><given-names>A</given-names> </name><name name-style="western"><surname>Van Alen</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Plagwitz</surname><given-names>L</given-names> </name><name name-style="western"><surname>Varghese</surname><given-names>J</given-names> </name></person-group><article-title>Classification of Parkinson&#x2019;s disease from voice - analysis of data selection bias</article-title><source>Stud Health Technol Inform</source><year>2023</year><month>05</month><day>18</day><volume>302</volume><fpage>127</fpage><lpage>128</lpage><pub-id pub-id-type="doi">10.3233/SHTI230079</pub-id><pub-id pub-id-type="medline">37203624</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ozbolt</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Moro-Velazquez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lina</surname><given-names>I</given-names> </name><name name-style="western"><surname>Butala</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Dehak</surname><given-names>N</given-names> </name></person-group><article-title>Things to consider when automatically detecting Parkinson&#x2019;s disease using the phonation of sustained vowels: analysis of methodological issues</article-title><source>Appl Sci</source><year>2022</year><volume>12</volume><issue>3</issue><fpage>991</fpage><pub-id pub-id-type="doi">10.3390/app12030991</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Quamar</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ambeth Kumar</surname><given-names>VD</given-names> </name><name name-style="western"><surname>Rizwan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bagdasar</surname><given-names>O</given-names> </name><name name-style="western"><surname>Kadar</surname><given-names>M</given-names> </name></person-group><article-title>Voice-based early diagnosis of Parkinson&#x2019;s disease using spectrogram features and AI models</article-title><source>Bioengineering (Basel)</source><year>2025</year><month>09</month><day>29</day><volume>12</volume><issue>10</issue><fpage>1052</fpage><pub-id pub-id-type="doi">10.3390/bioengineering12101052</pub-id><pub-id pub-id-type="medline">41155050</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rahman</surname><given-names>W</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Islam</surname><given-names>MS</given-names> </name><etal/></person-group><article-title>Detecting Parkinson disease using a web-based speech task: observational study</article-title><source>J Med Internet Res</source><year>2021</year><month>10</month><day>19</day><volume>23</volume><issue>10</issue><fpage>e26305</fpage><pub-id pub-id-type="doi">10.2196/26305</pub-id><pub-id pub-id-type="medline">34665148</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paddo</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Purkayastha</surname><given-names>S</given-names> </name><name name-style="western"><surname>Newsome</surname><given-names>J</given-names> </name><name name-style="western"><surname>Trivedi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Gichoya</surname><given-names>JW</given-names> </name></person-group><article-title>Innovating challenges and experiences in Emory Health AI Bias Datathon: experience report</article-title><source>J Imaging Inform Med</source><year>2025</year><month>12</month><volume>38</volume><issue>6</issue><fpage>4293</fpage><lpage>4302</lpage><pub-id pub-id-type="doi">10.1007/s10278-024-01367-5</pub-id><pub-id pub-id-type="medline">40000544</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="web"><article-title>Datasets</article-title><source>Bridge2AI</source><access-date>2026-07-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://bridge2ai.org/datasets/">https://bridge2ai.org/datasets/</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Bensoussan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sigaras</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rameau</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Bridge2AI-voice: an ethically-sourced, diverse voice dataset linked to health information</article-title><source>PhysioNet</source><access-date>2026-07-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://physionet.org/content/b2ai-voice/3.0.0/">https://physionet.org/content/b2ai-voice/3.0.0/</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Churchwell</surname><given-names>C</given-names> </name><name name-style="western"><surname>Morrison</surname><given-names>M</given-names> </name><name name-style="western"><surname>Pardo</surname><given-names>B</given-names> </name></person-group><article-title>High-fidelity neural phonetic posteriorgrams</article-title><conf-name>2024 IEEE International Conference on Acoustics, Speech, and Signal Processing Workshops (ICASSPW)</conf-name><conf-date>Apr 14-19, 2024</conf-date><conf-loc>Seoul, Korea, Republic of</conf-loc><fpage>823</fpage><lpage>827</lpage><pub-id pub-id-type="doi">10.1109/ICASSPW62465.2024.10669905</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Shukla</surname><given-names>S</given-names> </name></person-group><article-title>Bridge2ai-voice-parkinsons-ast</article-title><source>GitHub</source><access-date>2026-7-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/Amorfati123/bridge2ai-voice-parkinsons-ast">https://github.com/Amorfati123/bridge2ai-voice-parkinsons-ast</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hire&#x0161;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Drot&#x00E1;r</surname><given-names>P</given-names> </name><name name-style="western"><surname>Pah</surname><given-names>ND</given-names> </name><name name-style="western"><surname>Ngo</surname><given-names>QC</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>DK</given-names> </name></person-group><article-title>On the inter-dataset generalization of machine learning approaches to Parkinson&#x2019;s disease detection from voice</article-title><source>Int J Med Inform</source><year>2023</year><month>11</month><volume>179</volume><fpage>105237</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2023.105237</pub-id><pub-id pub-id-type="medline">37801807</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Banerjee</surname><given-names>I</given-names> </name><name name-style="western"><surname>Bhattacharjee</surname><given-names>K</given-names> </name><name name-style="western"><surname>Burns</surname><given-names>JL</given-names> </name><etal/></person-group><article-title>&#x201C;Shortcuts&#x201D; causing bias in radiology artificial intelligence: causes, evaluation, and mitigation</article-title><source>J Am Coll Radiol</source><year>2023</year><month>09</month><volume>20</volume><issue>9</issue><fpage>842</fpage><lpage>851</lpage><pub-id pub-id-type="doi">10.1016/j.jacr.2023.06.025</pub-id><pub-id pub-id-type="medline">37506964</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Demographic bias of expert-level vision-language foundation models in medical imaging</article-title><source>Sci Adv</source><year>2025</year><month>03</month><day>28</day><volume>11</volume><issue>13</issue><pub-id pub-id-type="doi">10.1126/sciadv.adq0305</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Orozco-Arroyave</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Arias-Londo&#x00F1;o</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Vargas-Bonilla</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Gonz&#x00E1;lez-R&#x00E1;tiva</surname><given-names>MC</given-names> </name><name name-style="western"><surname>N&#x00F6;th</surname><given-names>E</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Calzolari</surname><given-names>N</given-names> </name><name name-style="western"><surname>Choukri</surname><given-names>K</given-names> </name><name name-style="western"><surname>Declerck</surname><given-names>T</given-names> </name></person-group><article-title>New spanish speech corpus database for the analysis of people suffering from parkinson&#x2019;s disease</article-title><year>2014</year><access-date>2026-03-17</access-date><conf-name>Ninth International Conference on Language Resources and Evaluation</conf-name><conf-date>May 26-31, 2014</conf-date><conf-loc>Reykjavik, Iceland</conf-loc><fpage>342</fpage><lpage>347</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/L14-1549/">https://aclanthology.org/L14-1549/</ext-link></comment></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sakar</surname><given-names>BE</given-names> </name><name name-style="western"><surname>Isenkul</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Sakar</surname><given-names>CO</given-names> </name><etal/></person-group><article-title>Collection and analysis of a Parkinson speech dataset with multiple types of sound recordings</article-title><source>IEEE J Biomed Health Inform</source><year>2013</year><month>07</month><volume>17</volume><issue>4</issue><fpage>828</fpage><lpage>834</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2013.2245674</pub-id><pub-id pub-id-type="medline">25055311</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wroge</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Ozkanca</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Demiroglu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Si</surname><given-names>D</given-names> </name><name name-style="western"><surname>Atkins</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Ghomi</surname><given-names>RH</given-names> </name></person-group><article-title>Parkinson&#x2019;s disease diagnosis using machine learning and voice</article-title><year>2018</year><conf-name>2018 IEEE Signal Processing in Medicine and Biology Symposium (SPMB)</conf-name><conf-date>Dec 1, 2018</conf-date><conf-loc>Philadelphia, PA</conf-loc><fpage>1</fpage><lpage>7</lpage><pub-id pub-id-type="doi">10.1109/SPMB.2018.8615607</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tracy</surname><given-names>JM</given-names> </name><name name-style="western"><surname>&#x00D6;zkanca</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Atkins</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Hosseini Ghomi</surname><given-names>R</given-names> </name></person-group><article-title>Investigating voice as a biomarker: Deep phenotyping methods for early detection of Parkinson&#x2019;s disease</article-title><source>J Biomed Inform</source><year>2020</year><month>04</month><volume>104</volume><fpage>103362</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2019.103362</pub-id><pub-id pub-id-type="medline">31866434</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amato</surname><given-names>F</given-names> </name><name name-style="western"><surname>Borz&#x00EC;</surname><given-names>L</given-names> </name><name name-style="western"><surname>Olmo</surname><given-names>G</given-names> </name><name name-style="western"><surname>Orozco-Arroyave</surname><given-names>JR</given-names> </name></person-group><article-title>An algorithm for Parkinson&#x2019;s disease speech classification based on isolated words analysis</article-title><source>Health Inf Sci Syst</source><year>2021</year><month>12</month><volume>9</volume><issue>1</issue><fpage>32</fpage><pub-id pub-id-type="doi">10.1007/s13755-021-00162-8</pub-id><pub-id pub-id-type="medline">34422258</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jeong</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>EC</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>HJ</given-names> </name></person-group><article-title>Exploring spectrogram-based audio classification for Parkinson&#x2019;s disease: a study on speech classification and qualitative reliability verification</article-title><source>Sensors (Basel)</source><year>2024</year><month>07</month><day>17</day><volume>24</volume><issue>14</issue><fpage>4625</fpage><pub-id pub-id-type="doi">10.3390/s24144625</pub-id><pub-id pub-id-type="medline">39066023</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Piao</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kemps</surname><given-names>H</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>T</given-names> </name><name name-style="western"><surname>Saeed</surname><given-names>A</given-names> </name></person-group><article-title>Unified acoustic representations for screening neurological and respiratory pathologies from voice</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 19, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2508.20717</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Luz</surname><given-names>S</given-names> </name><name name-style="western"><surname>Haider</surname><given-names>F</given-names> </name><name name-style="western"><surname>Fuente</surname><given-names>S de la</given-names> </name><name name-style="western"><surname>Fromm</surname><given-names>D</given-names> </name><name name-style="western"><surname>MacWhinney</surname><given-names>B</given-names> </name></person-group><article-title>Alzheimer&#x2019;s dementia recognition through spontaneous speech: the ADReSS challenge</article-title><conf-name>Interspeech 2020</conf-name><conf-date>Oct 25-29, 2020</conf-date><conf-loc>Shanghai, China</conf-loc><fpage>2172</fpage><lpage>2176</lpage><pub-id pub-id-type="doi">10.21437/Interspeech.2020-2571</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Detailed preprocessing and model training procedures, baseline and ablation comparisons, calibration and threshold-sensitivity analyses, per-fold and per-task results, spectrogram versus phonetic posteriorgram foundation model comparison, a supplementary depression screening analysis, attention map analysis, and a survey of demographic reporting in voice-based Parkinson disease classification studies.</p><media xlink:href="jmir_v28i1e95609_app1.docx" xlink:title="DOCX File, 3109 KB"/></supplementary-material><supplementary-material id="app2"><label>Checklist 1</label><p>TRIPOD+AI checklist.</p><media xlink:href="jmir_v28i1e95609_app2.docx" xlink:title="DOCX File, 35 KB"/></supplementary-material></app-group></back></article>