<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e96347</article-id><article-id pub-id-type="doi">10.2196/96347</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Multilingual Evidence-Based Question-Answering for Stroke Discharge Summaries: Study of Cross-Lingual Heterogeneity in Clinical Reports</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Lanz</surname><given-names>Vojt&#x011B;ch</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Datseris</surname><given-names>Aleksis</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Boytcheva</surname><given-names>Svetla</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mayer</surname><given-names>Ji&#x0159;&#x00ED;</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zik&#x00E1;nov&#x00E1;</surname><given-names>&#x0160;&#x00E1;rka</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mikul&#x00ED;k</surname><given-names>Robert</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pecina</surname><given-names>Pavel</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Institute of Formal and Applied Linguistics, Faculty of Mathematics and Physics, Charles University</institution><addr-line>Malostransk&#x00E9; n&#x00E1;m&#x011B;st&#x00ED; 25</addr-line><addr-line>Prague</addr-line><country>Czech Republic</country></aff><aff id="aff2"><institution>Graphwise</institution><addr-line>Sofia</addr-line><country>Bulgaria</country></aff><aff id="aff3"><institution>Computer Informatics, Faculty of Mathematics and Informatics, Sofia University "St. Kliment Ohridski"</institution><addr-line>Sofia</addr-line><country>Bulgaria</country></aff><aff id="aff4"><institution>Department of Linguistic Modelling and Knowledge Processing, Faculty of Mathematics and Informatics, Sofia University "St. Kliment Ohridski"</institution><addr-line>Sofia</addr-line><country>Bulgaria</country></aff><aff id="aff5"><institution>Health Management Institute</institution><addr-line>Brno</addr-line><country>Czech Republic</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Ogunbowale</surname><given-names>Oluwatobilola</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Liu</surname><given-names>Zhao</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Vojt&#x011B;ch Lanz, MSc, Institute of Formal and Applied Linguistics, Faculty of Mathematics and Physics, Charles University, Malostransk&#x00E9; n&#x00E1;m&#x011B;st&#x00ED; 25, Prague, 118 00, Czech Republic, 420 951 554 278; <email>lanz@ufal.mff.cuni.cz</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>19</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e96347</elocation-id><history><date date-type="received"><day>08</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>18</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>19</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Vojt&#x011B;ch Lanz, Aleksis Datseris, Svetla Boytcheva, Ji&#x0159;&#x00ED; Mayer, &#x0160;&#x00E1;rka Zik&#x00E1;nov&#x00E1;, Robert Mikul&#x00ED;k, Pavel Pecina. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 19.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e96347"/><abstract><sec><title>Background</title><p>The Registry of Stroke Care Quality (RES-Q) is a health care quality improvement platform used globally. RES-Q collects structured quality-of-care data for patients with stroke, requiring clinicians to manually extract information from electronic health records or documents such as discharge summaries. This process is essential but time-consuming, particularly given the variability, length, and semistructured nature of clinical reports.</p></sec><sec><title>Objective</title><p>This study aimed to develop and evaluate a multilingual Evidence-Based Question-Answering framework that identifies supporting text spans in clinical reports of patients with stroke and proposes answer suggestions for structured clinical forms, with the goal of reducing clinician workload while preserving full human oversight.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conduct a multilingual study using more than 1500 pseudonymized stroke discharge summaries in 5 languages, annotated with question-evidence-answer triplets. Encoder-based language models are used to extract evidence spans from the reports, while generative language models are used to predict normalized form answers based on the extracted evidences. We compare multiple training strategies, such as (1) models trained on reports in a single target language, (2) models trained jointly on reports in different languages, and (3) models trained on original reports combined with cross-lingual data augmentations. We evaluate performance on Evidence Extraction, Answer Prediction, and end-to-end Evidence-Based Question Answering across the 5 languages.</p></sec><sec sec-type="results"><title>Results</title><p>The presented Evidence-Based Question-Answering system achieves 88% end-to-end accuracy in form filling across 5 languages (77% for patient-specific questions and 95% for default or unverifiable items). Evidence Extraction is the primary bottleneck, reaching 85% <italic>F</italic><sub>1</sub> and 79% exact match, whereas Answer Prediction based on extracted evidences is more stable, achieving 95% accuracy. The performance varies by question type, and cross-lingual training generally reduces Evidence Extraction performance but has little effect on Answer Prediction. Model performance is influenced more by reporting practices and dataset characteristics than by language itself.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Evidence-Based Question Answering over multilingual stroke discharge summaries enables human-in-the-loop validation and effective answer prediction with moderate computational resources. Evidence Extraction is the main bottleneck, while Answer Prediction is robust across languages and model sizes. The approach supports structured data collection, although generalization to new languages requires target-language training data.</p></sec></abstract><kwd-group><kwd>clinical question answering</kwd><kwd>stroke discharge summaries</kwd><kwd>evidence extraction</kwd><kwd>multilingual clinical natural language processing</kwd><kwd>medical large language models</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Improving the quality of health care delivery saves lives. Across diverse medical domains, systematic quality monitoring and feedback have been shown to reduce mortality, complications, and unwarranted variation in care by identifying gaps between evidence-based recommendations and real-world practice [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Despite its proven effectiveness, large-scale adoption of quality improvement programs remains limited. A major barrier is not the lack of clinical guidelines or clinician willingness, but the practical burden of data collection [<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>As clinical documentation increasingly moves into digital formats, physicians typically compose records as unstructured text stored in hospital information systems [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Current quality registries and audit systems still rely on manual extraction of structured variables from these unstructured documents, which can be lengthy, sometimes several pages long, making the process time-consuming, error-prone, and difficult to scale [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref9">9</xref>]. Many clinicians review these digitized texts during personal time, contributing to overwork and burnout [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. As a result, quality monitoring is often delayed, incomplete, or abandoned, especially in health care systems facing workforce shortages [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. Automating information extraction from clinical documents could reduce health care workers&#x2019; workload. This is where natural language processing (NLP) methods can provide valuable assistance [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>However, clinical text is not standard &#x201C;natural language&#x201D; in the traditional sense. In addition to domain-specific jargon, Latin terms, and specialized terminology, discharge summaries often contain enumerations, tables, numerical values, abbreviations, and shorthand phrases. Moreover, because these documents are usually written quickly and under time pressure, they frequently contain typographical errors and inconsistencies. These semistructured, lengthy, and often noisy documents therefore pose unique challenges for NLP models [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>].</p><p>Whether relevant information appears in discharge summaries, progress notes, admission reports, or other parts of the electronic health record (EHR) varies widely across hospitals, countries, and clinical cultures. At the same time, clinical documents differ substantially in structure, level of detail, and terminology depending on the specialty, the individual clinician, and the institution [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>]. Anchoring automated extraction methods to a single document type risks limiting generalizability. Instead, the key requirement is the ability to robustly retrieve clinically relevant information from heterogeneous, unstructured clinical text, regardless of its specific format or origin.</p><p>The language used in such texts also varies by hospital, country, and individual physician [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. Despite ongoing efforts to construct domain-specific clinical NLP datasets to support clinical NLP research, the sensitive nature of clinical data and patient privacy concerns severely limit the availability of publicly accessible resources, which are predominantly in English when available [<xref ref-type="bibr" rid="ref24">24</xref>]. As a result, conducting research and developing NLP tools for non-English clinical data remains challenging, since model performance strongly depends on the availability of domain-specific training data [<xref ref-type="bibr" rid="ref25">25</xref>]. Consequently, various pretrained models have been published, including multilingual models pretrained on general text [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>], English clinical-domain models [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>], and multilingual models further pretrained on clinical data [<xref ref-type="bibr" rid="ref30">30</xref>]. Interestingly, multilingual pretraining can be more beneficial than domain-specific English pretraining even for clinical NLP tasks in English, for which both relevant training data and models already exist [<xref ref-type="bibr" rid="ref31">31</xref>].</p><p>Data protection and privacy are also limitations of manual data extraction, yet they are rarely discussed. Manual registry entry requires a physical person to read and process large volumes of sensitive patient documentation, often outside the original care context. From a legal and ethical perspective, this creates avoidable exposure of personal health data and increases the risk of unauthorized access, secondary use, or data leakage [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. Automation using NLP methods could help mitigate these risks.</p><p>However, for the same reason, the use of third-party NLP models, such as ChatGPT (OpenAI) [<xref ref-type="bibr" rid="ref34">34</xref>], is infeasible, since the data cannot be shared with external parties beyond the hospital&#x2019;s control. At the same time, hospitals typically lack the computational resources necessary to run or fine-tune large-scale language models [<xref ref-type="bibr" rid="ref35">35</xref>]. As a result, developing effective tools for hospital environments must account for both model size and input and output length limitations [<xref ref-type="bibr" rid="ref36">36</xref>], making the direct processing of entire multipage reports inefficient and technically impractical.</p><p>Previous research on clinical information extraction from unstructured EHR initially relied on rule-based methods, classical machine learning classifiers, and recurrent neural networks [<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref45">45</xref>], with subsequent studies successfully adopting encoder-based models like BERT (Bidirectional Encoder Representations from Transformers) for specific clinical extraction tasks [<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]. While these approaches can achieve high performance on well-defined, relatively unambiguous variables (such as detecting large-vessel occlusion or silent brain infarcts), their performance often drops when dealing with complex, heterogeneous, or implicitly expressed clinical attributes. To explore alternatives for these more challenging tasks, recent work has tested prompt-based large language model (LLM) inference without task-specific fine-tuning for document-level classification and schema-defined field extraction directly from discharge summaries [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref49">49</xref>]. However, the practical deployment of their findings may still be constrained by privacy requirements, computational resources, and the need to process lengthy clinical documents.</p><p>Until NLP models can provide guaranteed accuracy (which they currently cannot), their predictions cannot be relied upon blindly. This limitation is particularly critical in the clinical domain, where we might need to extract life-critical information, and is further complicated by the phenomenon of hallucinations in generative models [<xref ref-type="bibr" rid="ref50">50</xref>]. Current mitigation strategies include retrieval-augmented generation (RAG), in which the model is provided with relevant context retrieved from documents to improve the quality of generated answers, and human-in-the-loop (HITL) methods, where a human validator supervises the output [<xref ref-type="bibr" rid="ref51">51</xref>]. However, even with advanced RAG techniques, guaranteed accuracy cannot be achieved, while sole reliance on HITL methods is inefficient; if a human must manually verify every model prediction by searching the entire report for supporting evidence, they could effectively answer the question themselves, undermining the intended benefit of NLP-based assistance.</p><p>We propose to combine the strengths of both approaches. Our method generates answers to clinical questions based on clinical documents, while also providing the evidential spans in the text that support each answer. This facilitates more accurate RAG-style generation [<xref ref-type="bibr" rid="ref52">52</xref>-<xref ref-type="bibr" rid="ref56">56</xref>] and enables clinicians to quickly validate the model&#x2019;s output by referring directly to the relevant passages.</p><p>Identifying supporting evidence is nontrivial. Within a report, a question may have no text snippets that provide supporting evidence, a single concise span, or multiple complementary spans that need to be combined to determine the correct answer. The number of evidence spans, therefore, varies across questions and reports, making this real-world task more complex than conventional span-based Question-Answering reading comprehension setups [<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref58">58</xref>], where encoder-based models have traditionally achieved the strongest performance [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref60">60</xref>].</p><p>The goal of this work is to design, evaluate, and demonstrate the potential of NLP methods to assist in reviewing clinical texts in different languages, rather than directly replacing clinicians, by providing easily verifiable predictions that improve efficiency without compromising safety or interpretability.</p><p>In this study, we limit our analysis to multilingual discharge summaries related to patients with stroke and evaluate model performance on normalized questions whose answers contribute to the Registry of Stroke Care Quality (RES-Q) [<xref ref-type="bibr" rid="ref61">61</xref>]. RES-Q is a global quality improvement platform that relies on structured variables extracted from routine clinical records to assess and improve stroke care processes worldwide. These data are extracted from discharge summaries written in different languages, and physicians currently review lengthy reports manually to fill structured forms comprising over two hundred normalized questions with predefined answer types (eg, binary, numeric, multiple choice, date-based, and open-ended), creating a substantial workload and limiting scalability.</p><p>The experiments in this study are conducted on the resqEQA dataset (described in the Data section), which comprises 1500 pseudonymized discharge reports of patients with stroke across 5 well-represented languages. The dataset is annotated by clinical experts with RES-Q form responses (answers), along with the corresponding evidence spans in the text. To our knowledge, this is the first work to explore the application of NLP models to this dataset. We also provide an in-depth analysis of how multilinguality and local documentation practices influence model performance in the clinical domain. In particular, we investigate whether a single shared model can be used across all languages or if one model per language is preferable, how data augmentation into other languages impacts performance, whether trained models can be transferred to unseen languages, and, in general, how well this task can be addressed given these real-world constraints. We then decompose this Evidence-Based Question Answering task into two stages, which we explore separately, as visualized in <xref ref-type="fig" rid="figure1">Figure 1</xref>:</p><list list-type="order"><list-item><p>Evidence Extraction: locating text spans within the discharge summary that support the answer to a given question.</p></list-item><list-item><p>Answer Prediction: predicting the final answer based on the extracted evidence.</p></list-item></list><p>Overall, this work aims to develop and evaluate a multilingual Evidence-Based Question Answering framework that supports structured clinical data extraction from stroke discharge summaries while providing evidence spans in the original text across languages, allowing analysis of cross-lingual differences.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>The proposed Evidence-Based Question-Answering pipeline consisting of Evidence Extraction and Answer Prediction components.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96347_fig01.png"/></fig></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Data</title><p>For our experiments, we use the resqEQA dataset, an Evidence-Based Question-Answering dataset in the clinical domain of stroke. The dataset consists of discharge summaries, each paired with a set of question-evidences-answer triplets. Each question is designed for a structured form and therefore lies somewhere between a natural question and a form label. For each question, evidences denote a list of substrings extracted from the given discharge summary context, and the answer represents the final form-compliant response to the question, determined directly from the evidence list (which may contain 0, 1, or multiple elements). The dataset contains 5 question types: boolean (true or false), date-time (date, time, or both), enumeration (multiple-choice with a unique option set), integer, float, and open-ended string. The dataset covers 6 languages: Bulgarian, Greek, English, Spanish, Polish, and Romanian. In total, resqEQA comprises 1596 reports and 181,204 question-evidence-answer triplets, of which 115,203 are impossible cases, that is, questions that have an answer but no supporting text in the discharge summary can be associated with the question, meaning that the answer may be a default value or implicitly inferred from other context. The remaining triplets are possible cases, where at least 1 supporting evidence span is present in the report text. <xref ref-type="table" rid="table1">Table 1</xref> presents detailed statistics for each language, including the number of reports, total instances, and the split between possible (containing at least 1 evidence) and impossible (evidence list empty) instances. All languages except Spanish contain hundreds of reports; Spanish therefore serves only as a small reference test set, not sufficient for training.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Basic statistics of the resqEQA dataset across 6 different languages.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Language</td><td align="left" valign="bottom">Reports</td><td align="left" valign="bottom">QEA<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> triplets</td><td align="left" valign="bottom">Possible instances</td><td align="left" valign="bottom">Impossible instances</td></tr></thead><tbody><tr><td align="left" valign="top">Bulgarian</td><td align="left" valign="top">286</td><td align="left" valign="top">34,120</td><td align="left" valign="top">11,679</td><td align="left" valign="top">22,441</td></tr><tr><td align="left" valign="top">Greek</td><td align="left" valign="top">303</td><td align="left" valign="top">32,089</td><td align="left" valign="top">15,305</td><td align="left" valign="top">16,784</td></tr><tr><td align="left" valign="top">English</td><td align="left" valign="top">311</td><td align="left" valign="top">28,480</td><td align="left" valign="top">10,313</td><td align="left" valign="top">18,167</td></tr><tr><td align="left" valign="top">Spanish</td><td align="left" valign="top">27</td><td align="left" valign="top">3581</td><td align="left" valign="top">1007</td><td align="left" valign="top">2574</td></tr><tr><td align="left" valign="top">Polish</td><td align="left" valign="top">376</td><td align="left" valign="top">48,952</td><td align="left" valign="top">15,338</td><td align="left" valign="top">33,614</td></tr><tr><td align="left" valign="top">Romanian</td><td align="left" valign="top">293</td><td align="left" valign="top">33,982</td><td align="left" valign="top">12,359</td><td align="left" valign="top">21,623</td></tr><tr><td align="left" valign="top">Total</td><td align="left" valign="top">1596</td><td align="left" valign="top">181,204</td><td align="left" valign="top">66,001</td><td align="left" valign="top">115,203</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>QEA: question-evidence-answer.</p></fn></table-wrap-foot></table-wrap><p>On average, a single report contains more than 2000 words and nearly 17,000 characters. Additionally, there are structural differences in reports across languages. For example, Bulgarian reports contain fewer lines and paragraphs than other languages, while Spanish reports do not exhibit paragraph structure at all. Polish reports are considerably the longest, whereas English reports are notably the shortest.</p><p>A large proportion of question-evidence-answer instances are <italic>impossible</italic>. The full distribution of evidence counts per question per language is illustrated in <xref ref-type="fig" rid="figure2">Figure 2</xref>. The distribution follows a Poisson-like pattern: it is uncommon for a question to contain more than 1 or 2 pieces of evidence. However, when evidence for a given question is present, Polish and Spanish evidence segments are, on average, the longest. For Polish, this correlates with the fact that Polish reports are the longest and contain the densest information. Overall, for <italic>possible</italic> questions, an average of approximately 4 words from the report is required to answer a given question. For a detailed breakdown of both report lengths and evidence lengths, refer to <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Distribution of the number of evidence per question-evidences-answer instance across all 6 languages in the resqEQA dataset.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96347_fig02.png"/></fig><p>Across the entire <italic>resqEQA</italic> dataset, questions originate from the RES-Q form [<xref ref-type="bibr" rid="ref61">61</xref>], which contains 238 unique questions, most of which are optional (either irrelevant to the case or not required). Each report therefore includes each form question at most once (either exactly once, or not at all if the question was irrelevant and therefore not annotated). Questions fall into the following categories: anamnesis, onset, admission, diagnosis, treatment, postacute care, discharge, and postdischarge. Each question belongs to one of the predefined types: boolean, integer, number, date-time, enumeration, or open-ended string. The frequency distribution of question types across all reports and languages is presented in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, where boolean questions dominate, followed by enumeration and integer questions.</p><p><xref ref-type="fig" rid="figure3">Figure 3</xref> illustrates an example report text, showing a sample of predefined form questions with their answers, where the highlighted texts in the discharge summary indicate evidence mapped to the filled answers in the form.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Example of an annotated report paragraph, illustrating sample questions with their answers and supporting evidences. COPD: chronic obstructive pulmonary disorder; CT: computed tomography; ECG: electrocardiography; MR: mitral regurgitation; PMH: past medical history; SVT: supraventricular tachycardia; TR: tricuspid regurgitation; USS: ultrasound scan.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96347_fig03.png"/></fig><p>For our experiments, we used a random split of the train, development, and test subsets for each language such that the test set contains 60 reports per language, the development set contains 30 reports per language, and the remaining reports are placed in the training set. The only exception is Spanish, which is fully included in the test set; therefore, Spanish is not included in the main experiments and is used only as a reference for zero-shot languages.</p><p>Evidence and answers for the reports were annotated by clinicians with clinical expertise from different countries on real pseudonymized clinical discharge summary cases. All annotators were provided with identical multipage annotation guidelines that deterministically specified the annotation procedure. All annotators also had access to the same set of questions from a shared questionnaire, which was only translated for specific languages when necessary. Due to the time-consuming nature of the task and the limited availability of physicians, each report was annotated by only 1 clinician. However, interannotator agreement was measured on 10 reports across different languages using 2 independent annotators. On 443 question instances annotated by both annotators, the agreement corresponds to 92.1% accuracy in end-to-end question-answering performance over the final answers. For an additional 139 questions (24% of all annotated questions), annotators disagreed on relevance, meaning these were completed by only 1 of the 2 annotators.</p></sec><sec id="s2-2"><title>Task Definitions and Evaluation Metrics</title><p>The resqEQA dataset comprises instances of question-evidence-answer triplets. More specifically, each instance contains a report text, a question, a list of supporting evidence, the corresponding form answer, and the list of possible form options. Due to the sensitive and potentially life-critical nature of clinical data, it is important not only to predict the correct form answer but also to identify the precise evidence supporting that answer, enabling clinicians at HITL to easily validate the prediction. In addition, we aim to evaluate how accurately form answers can be predicted based solely on the extracted evidence, as well as to assess the overall performance of the end-to-end prediction pipeline. Accordingly, we define 3 tasks for evaluation, referred to as S1, S2, and S1-S2, where S1 and S2 correspond to the first (Evidence Extraction) and second stage (Answer Prediction) of the full pipeline, respectively, and S1-S2 denotes the end-to-end pipeline combining both stages:</p><sec id="s2-2-1"><title>S1: Evidence Extraction</title><p>Given a pair consisting of a report text (as context) and a question, the task is to identify the minimal list of concise substrings from the report that collectively serve as supporting evidence for answering the question, such that no additional substrings in the report serve as evidence. Correctness is evaluated against the gold evidence list by first concatenating all substrings in the order they appear in the report context, and then comparing the predicted and gold concatenated strings. Evaluation is performed using token-level <italic>F</italic><sub>1</sub> and exact match (EM) scores following the SQuAD evaluation script [<xref ref-type="bibr" rid="ref57">57</xref>]. Tokens from the predicted and gold strings are compared, and <italic>F</italic><sub>1</sub> reflects the overlap. If the gold evidence list is empty, a prediction stating that there are no evidence spans is scored as <italic>F</italic><sub>1</sub> 100%, otherwise 0%. EM is computed as a strict string comparison: 100% if the predicted and gold strings are identical, and 0% if they differ in any character.</p></sec><sec id="s2-2-2"><title>S2: Answer Prediction</title><p>Given the already correct gold list of evidences, the question, and the set of possible answers, the task is to predict the correct form answer. Accuracy is measured as the proportion of instances where the predicted answer matches the gold answer exactly.</p></sec><sec id="s2-2-3"><title>S1-S2: Evidence-Based Question Answering</title><p>The full Question-Answering pipeline, combining the two previous tasks, is evaluated end-to-end. Given the report context, the question, and the set of possible answers as input, the goal is to predict the correct form answer. Evaluation is performed using accuracy, in the same manner as for the Answer-Prediction task, measured as the proportion of instances where the predicted answer exactly matches the gold answer.</p></sec></sec><sec id="s2-3"><title>Proposed Approach</title><p>This section presents the methodology proposed in this work for addressing the real-world Evidence-Based Question Answering (S1-S2) task, implemented as a 2-stage pipeline combining Evidence Extraction (S1) and Answer Prediction (S2) for end-to-end answer prediction.</p></sec><sec id="s2-4"><title>Evidence Extraction (S1)</title><p>Question Answering as an NLP task can take several forms. One variant is comprehensive reading span-based question answering, where the objective is to locate a span within the provided context that answers a given question. Datasets such as SQuAD [<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref62">62</xref>] are commonly used for this type of task, containing paragraph contexts with either 0 or 1 answer span. These tasks are typically addressed using encoder-based models [<xref ref-type="bibr" rid="ref63">63</xref>-<xref ref-type="bibr" rid="ref66">66</xref>], such as BERT (Google AI) and its variants [<xref ref-type="bibr" rid="ref26">26</xref>].</p><p>Previous works [<xref ref-type="bibr" rid="ref67">67</xref>] approached span prediction by adding 2 layers on top of the encoder architecture: one for predicting the beginning of the span and another for predicting the end, both trained with cross-entropy loss. If no evidence span exists in the provided context, both the start and end positions are set to the [CLS] token, a special token used in transformer-based language models that marks the beginning of the sequence rather than containing any input content and acts as a summary representation of the entire input sequence [<xref ref-type="bibr" rid="ref67">67</xref>]. During inference, all valid (start and end) pairs are considered (with start&#x003C;end, including the [CLS]-[CLS] pair). The score for each candidate span is calculated by multiplying the softmax probabilities of the start and end indices from the 2 layers. The span with the highest score is selected. If the [CLS]-[CLS] pair is chosen, it indicates that no evidence is present in the context for the given question.</p><p>Although encoder-based models have become less prevalent compared with generative LLMs, it is crucial to select models based on task suitability rather than prevailing trends. In the present task, the objective is to identify specific evidence spans rather than to generate novel content; the goal is to locate precise textual pointers within existing reports. Accordingly, encoder-based models generally demonstrate higher reliability and performance for this type of task [<xref ref-type="bibr" rid="ref63">63</xref>-<xref ref-type="bibr" rid="ref66">66</xref>,<xref ref-type="bibr" rid="ref68">68</xref>]. Furthermore, generative LLMs require substantially more computational resources and longer inference times, which limits their practical applicability for this task.</p><p>Therefore, in our evidence extraction (S1) task, we focus on encoder-based methods. However, long clinical reports pose a challenge, as classical pretrained BERT models are limited to 512 tokens [<xref ref-type="bibr" rid="ref26">26</xref>], and even long-context models, such as LongFormer (Allen Institute for Artificial Intelligence) [<xref ref-type="bibr" rid="ref69">69</xref>] or BigBird (Google Research) [<xref ref-type="bibr" rid="ref70">70</xref>], support only up to 4096 tokens, which is still insufficient in our setting (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). In addition, reports can contain multiple evidence spans. To address this, we split the report text into chunks of 256 tokens, with a 32-token overlap between neighboring chunks. This overlap prevents evidence from being split across chunk boundaries. Chunking also ensures that most segments contain either 0 or 1 evidence span, which can then be processed using standard methods, as adapted for tasks such as SQuAD, described at the beginning of this section.</p><p>But still, some chunks may still contain multiple evidences. To handle these cases, we propose a new iterative method as visualized in <xref ref-type="fig" rid="figure4">Figure 4</xref>. The model first identifies the most probable span by maximizing the probability of its start and end positions, comparing it against the probability of the [CLS]-[CLS] span, which indicates no evidence. If the identified span has a probability higher than [CLS]-[CLS], it is stored, masked with a special token, and the process is repeated iteratively until the [CLS]-[CLS] span becomes the most probable prediction. Neighboring or overlapping spans from the same or different chunks are then merged to form complete evidence spans, resulting in 0, 1, or multiple evidence per question and report.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Iterative Evidence Extraction (S1) pipeline for a single context segment and question, repeatedly predicting evidences until the [CLS]&#x2013;[CLS] span is returned by the model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96347_fig04.png"/></fig><p>We experiment with and compare general-domain multilingual models, including Multilingual BERT (mBERT) [<xref ref-type="bibr" rid="ref26">26</xref>], XLM-Roberta (XLMR) [<xref ref-type="bibr" rid="ref71">71</xref>], and Multilingual ModernBERT (mmBERT) [<xref ref-type="bibr" rid="ref72">72</xref>], selected for their combination of multilingual coverage, established strong performance on span-based question-answering tasks, and availability as pretrained encoders. To the best of our knowledge, no publicly available encoder-based models are pretrained for both clinical and multilingual data. Furthermore, previous studies have shown that general-domain multilingual models can outperform clinically pretrained English models on several clinical English tasks [<xref ref-type="bibr" rid="ref31">31</xref>]. Therefore, for English reports in resqEQA, we additionally compare the English ClinicalBERT [<xref ref-type="bibr" rid="ref28">28</xref>] with its multilingual nonclinical counterpart, mBERT, and similarly compare ClinicalModernBERT [<xref ref-type="bibr" rid="ref29">29</xref>] with mmBERT.</p></sec><sec id="s2-5"><title>Answer Prediction (S2)</title><p>For the Answer Prediction (S2) task, we use the generative models Llama3-8B (Meta AI) [<xref ref-type="bibr" rid="ref27">27</xref>], Mistral-7B-Instruct-v0.1 (Mistral AI) [<xref ref-type="bibr" rid="ref73">73</xref>], Phi-3.5-mini-instruct 3.8B (Microsoft) [<xref ref-type="bibr" rid="ref74">74</xref>], and Gemma3-4B (Google DeepMind) [<xref ref-type="bibr" rid="ref75">75</xref>], and compare each with its clinically oriented counterpart Med42-8B (M42 Health AI Team) [<xref ref-type="bibr" rid="ref30">30</xref>], BioMistral-8B [<xref ref-type="bibr" rid="ref76">76</xref>], MediPhi [<xref ref-type="bibr" rid="ref77">77</xref>], and MedGemma-4B-IT [<xref ref-type="bibr" rid="ref78">78</xref>], respectively. These models were selected as the currently available generative models with medical pretraining up to 8B parameters, including the base models used for clinical pretraining, offering a balance of state-of-the-art performance and practical computational feasibility. The objective of the Answer Prediction (S2) is to generate the final answer in its exact predefined normalized form, given the question, the corresponding set of possible answers, and the set of evidences previously extracted from the report context. However, when evaluating this task independently, we rely on gold-annotated evidences instead, both to assess the upper bound of model performance for this component and to simulate scenarios in which a human annotator at HITL has already corrected or verified the predicted evidence list.</p><p>We compare the models first in few-shot (using additional examples of the same question type from the training data in the same language as the target question) and second in fine-tuned settings. The prompt used for Answer Prediction (S2), illustrated in <xref ref-type="fig" rid="figure5">Figure 5</xref>, is adapted to the type of the question, whether the required output corresponds to one of the predefined categorical options (A, B, C, &#x2026;), a boolean value (Yes or No), a numerical value, or another format specified by the form definition. When the question definition allows &#x201C;None&#x201D; as a valid option, the prompt also provides the model with the option to answer, &#x201C;I don&#x2019;t know.&#x201D;</p><p>Finetuning is performed using LoRA [<xref ref-type="bibr" rid="ref79">79</xref>] with hyperparameters &#x03B1;=16, <italic>lora_dropout</italic>=0.1, and <italic>r</italic>=64, focusing on the layers &#x201C;q_proj,&#x201D; &#x201C;k_proj,&#x201D; &#x201C;v_proj,&#x201D; &#x201C;o_proj,&#x201D; and &#x201C;out_proj,&#x201D; with a batch size of 6. Prompts are constrained to a maximum of 512 tokens; in cases where the evidence context is extremely long, it is truncated to fit within this limit. Additionally, 32 tokens are reserved for special tokens and the predicted answer in the final input sequence.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Prompt template for Answer Prediction (S2) given the evidence as context, the question, and the set of possible answers defined by the form specification.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96347_fig05.png"/></fig></sec><sec id="s2-6"><title>Evidence-Based Question Answering (S1-S2)</title><p>The performance of the full pipeline is evaluated within the RAG framework, as illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>, by sequentially combining Evidence Extraction (S1) and Answer Prediction (S2). In this setup, Answer Prediction (S2) relies on evidences predicted by the Evidence Extraction (S1) component rather than gold annotations. Specifically, the best-performing Evidence Extraction model is used to generate the list of predicted evidences, which is then fed into the best-performing Answer Prediction model to produce the final answer, assuming no human intervention at any stage. This procedure demonstrates the end-to-end capability of the pipeline. However, for Answer Prediction (S2), although the model is evaluated on predicted evidences to simulate real-world conditions, it is trained using gold evidences.</p></sec><sec id="s2-7"><title>Training Configurations</title><p>In our experiments, we compare different strategies for combining monolingual and multilingual training. Let the <italic>target language</italic> denote the language being tested, and all remaining languages from the resqEQA dataset are referred to as <italic>other languages</italic>. We then investigate whether a model for the target language performs better when trained solely on that language or additionally on other languages. For the latter, we use either augmented reports automatically translated via machine translation into other languages from the original target-language reports (following the augmentation pipeline of Lanz and Pecina [<xref ref-type="bibr" rid="ref31">31</xref>]) or the original reports in other languages themselves.</p><p>We also aim to investigate how variations in reports affect model performance. These variations may arise not only from differences in language but also from differences in country of origin. While multilingual augmentation has been shown to be beneficial in several tasks [<xref ref-type="bibr" rid="ref80">80</xref>,<xref ref-type="bibr" rid="ref81">81</xref>], it remains unclear whether training on reports from different countries, even in different languages, may introduce inconsistencies due to structural differences, writing style, or local conventions.</p><p>In addition, we are interested in determining how essential the original reports in the target language are, and whether it is possible to support new languages in a zero-shot setting for which we have no training data.</p><p>Therefore, we describe the different sets of training data whose combinations we explore (specifically those combinations described in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>):</p><list list-type="bullet"><list-item><p>Target original, <italic><bold>T</bold></italic>: Original reports written in the target language. For instance, when Bulgarian is the target language under test, only the original Bulgarian training reports are used.</p></list-item><list-item><p>Other original, <italic><bold>O</bold></italic>: Original reports written in all other languages. For instance, when Bulgarian is the target language under test, we use the original Greek, English, Polish, and Romanian training reports.</p></list-item><list-item><p>Target-to-other augmented, <bold><inline-formula><mml:math id="ieqn1"><mml:mi mathvariant="bold-italic">T</mml:mi><mml:mover><mml:mrow><mml:mo>&#x2192;</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">a</mml:mi><mml:mi mathvariant="bold-italic">u</mml:mi><mml:mi mathvariant="bold-italic">g</mml:mi></mml:mrow></mml:mover><mml:mi mathvariant="bold-italic">O</mml:mi></mml:math></inline-formula></bold>: Reports originally written in the target language, augmented into all other languages. For instance, when Bulgarian is the target language under test, we use augmented training reports translated from the original Bulgarian reports into Greek, English, Spanish, Polish, and Romanian.</p></list-item><list-item><p>Other-to-target augmented, <inline-formula><mml:math id="ieqn2"><mml:mi mathvariant="bold-italic">O</mml:mi><mml:mover><mml:mrow><mml:mo>&#x2192;</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">a</mml:mi><mml:mi mathvariant="bold-italic">u</mml:mi><mml:mi mathvariant="bold-italic">g</mml:mi></mml:mrow></mml:mover><mml:mi mathvariant="bold-italic">T</mml:mi></mml:math></inline-formula>: Reports originally written in other languages, augmented into the target language. For instance, when Bulgarian is the target language under test, we use augmented training reports translated from original reports in Greek, English, Polish, and Romanian into Bulgarian.</p></list-item><list-item><p>Other-to-other augmented, <inline-formula><mml:math id="ieqn3"><mml:mi mathvariant="bold-italic">O</mml:mi><mml:mover><mml:mrow><mml:mo>&#x2192;</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">a</mml:mi><mml:mi mathvariant="bold-italic">u</mml:mi><mml:mi mathvariant="bold-italic">g</mml:mi></mml:mrow></mml:mover><mml:mi mathvariant="bold-italic">O</mml:mi></mml:math></inline-formula>: Reports originally written in other languages, augmented into other languages. For instance, when Bulgarian is the target language under test, we use augmented training reports translated from original reports in English, Greek, Polish, and Romanian into Greek, English, Spanish, Polish, and Romanian.</p></list-item></list><p>Unless stated otherwise, results for these settings are reported for a single run per configuration due to the large number of experiments across multiple target languages and their computational cost. Model performance is averaged across the available languages to obtain a more stable estimate, reflecting the overall consistency and reliability of the results, while outcomes are also reported separately for each language.</p></sec><sec id="s2-8"><title>Ethical Considerations</title><p>The study protocol was reviewed and approved by the Ethics Committee of the Faculty of Mathematics and Physics, Charles University (approval REC260616). All data used in this study were handled in accordance with applicable ethical and data protection regulations. The dataset was pseudonymized before analysis, and no personally identifiable information was available to the researchers.</p><p>The annotation process was guided by predefined clinical guidelines intended to standardize labeling. Given the intended clinical use case, the proposed Evidence-Based Question-Answering (S1-S2) system is designed to operate in a HITL setting, as fully automated use without clinician validation could propagate extraction errors into structured clinical registry entries. However, to evaluate the theoretical impact of skipping this validation step, it is important to contextualize how registry data are used. Unlike systems designed for real-time, direct clinical decision-making for individual patients, platforms like RES-Q serve as tools for retrospective quality auditing and aggregate statistical monitoring. Because these clinical registries focus on large-scale data aggregation, any potentially unvalidated model errors would primarily impact systemic quality conclusions only if the model introduced systematic, nonrandom biases; randomly distributed errors tend to cancel out in the aggregate metrics. Furthermore, even manual data abstraction by clinicians is inherently prone to human error and rarely achieves perfect accuracy [<xref ref-type="bibr" rid="ref82">82</xref>]. While fully automated deployment remains unacceptable for safety-critical environments where maximum data integrity is required, the actual clinical safety footprint of isolated extraction errors in this specific setting is minimal.</p><p>Despite all languages using the same annotation guidelines and the fact that the questions in the form are direct translations that should not introduce bias, clinician annotators work with report text that follows different writing conventions in each language by different clinical annotators, which may significantly influence model performance.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Evidence Extraction (S1)</title><p>An important question is whether a single shared model trained jointly on all languages using all original training reports (<italic>T+O</italic>) is sufficient, or whether each language requires a separately trained model using only reports written in that target language (<italic>T</italic>). Another important aspect is the extent to which target-language reports are necessary for training and how well the model performs without them (<italic>O</italic>). Therefore, we compare these training settings in <xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>, where we report <italic>F</italic><sub>1</sub> and EM scores, respectively.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Evidence Extraction (S1) <italic>F</italic><sub>1</sub> performance across different models under various training configurations.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">T<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="bottom">O<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="bottom">Bulgarian</td><td align="left" valign="bottom">Greek</td><td align="left" valign="bottom">English</td><td align="left" valign="bottom">Polish</td><td align="left" valign="bottom">Romanian</td><td align="left" valign="bottom">Average</td><td align="left" valign="bottom">Spanish<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">XLMR<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content></td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">85.85</td><td align="left" valign="top">86.62</td><td align="left" valign="top">75.85</td><td align="left" valign="top">81.18</td><td align="left" valign="top">83.75</td><td align="left" valign="top">82.65</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">83.71</td><td align="left" valign="top">82.39</td><td align="left" valign="top">78.34</td><td align="left" valign="top">74.93</td><td align="left" valign="top">78.30</td><td align="left" valign="top">79.53</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">66.91</td><td align="left" valign="top">51.77</td><td align="left" valign="top">63.41</td><td align="left" valign="top">68.19</td><td align="left" valign="top">63.43</td><td align="left" valign="top">62.74</td><td align="char" char="." valign="top">68.82</td></tr><tr><td align="left" valign="top">mBERT<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">88.86</td><td align="left" valign="top">87.61</td><td align="left" valign="top">79.46</td><td align="left" valign="top">80.71</td><td align="left" valign="top">82.52</td><td align="left" valign="top">83.83</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">85.25</td><td align="left" valign="top">78.20</td><td align="left" valign="top">76.39</td><td align="left" valign="top">74.38</td><td align="left" valign="top">74.48</td><td align="left" valign="top">77.74</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">61.37</td><td align="left" valign="top">54.80</td><td align="left" valign="top">63.65</td><td align="left" valign="top">67.88</td><td align="left" valign="top">62.95</td><td align="left" valign="top">62.13</td><td align="char" char="." valign="top">71.91</td></tr><tr><td align="left" valign="top">mmBERT<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">90.53</td><td align="left" valign="top">90.12</td><td align="left" valign="top">80.72</td><td align="left" valign="top">81.66</td><td align="left" valign="top">80.80</td><td align="left" valign="top">84.77</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">87.45</td><td align="left" valign="top">83.90</td><td align="left" valign="top">79.97</td><td align="left" valign="top">75.24</td><td align="left" valign="top">76.71</td><td align="left" valign="top">80.65</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">66.76</td><td align="left" valign="top">55.16</td><td align="left" valign="top">63.48</td><td align="left" valign="top">66.50</td><td align="left" valign="top">63.38</td><td align="left" valign="top">63.06</td><td align="char" char="." valign="top">66.81</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Target language.</p></fn><fn id="table2fn2"><p><sup>b</sup>Other language.</p></fn><fn id="table2fn3"><p><sup>c</sup>Not included in the average and is intended only as an additional reference (its test set size differs from the others).</p></fn><fn id="table2fn4"><p><sup>d</sup>XLMR: XLM-Roberta.</p></fn><fn id="table2fn5"><p><sup>e</sup>mBERT: Multilingual BERT (Bidirectional Encoder Representations from Transformers).</p></fn><fn id="table2fn6"><p><sup>f</sup>mmBERT: Multilingual ModernBERT.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Evidence Extraction (S1) exact match performance across different models under various training configurations.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">T<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom">O<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="bottom">Bulgarian</td><td align="left" valign="bottom">Greek</td><td align="left" valign="bottom">English</td><td align="left" valign="bottom">Polish</td><td align="left" valign="bottom">Romanian</td><td align="left" valign="bottom">Average</td><td align="left" valign="bottom">Spanish<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">XLMR<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content></td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">80.54</td><td align="left" valign="top">79.95</td><td align="left" valign="top">69.67</td><td align="left" valign="top">74.61</td><td align="left" valign="top">77.17</td><td align="left" valign="top">76.39</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">78.88</td><td align="left" valign="top">75.28</td><td align="left" valign="top">72.59</td><td align="left" valign="top">69.61</td><td align="left" valign="top">72.18</td><td align="left" valign="top">73.71</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">66.15</td><td align="left" valign="top">48.92</td><td align="left" valign="top">63.12</td><td align="left" valign="top">66.78</td><td align="left" valign="top">62.47</td><td align="left" valign="top">61.49</td><td align="char" char="." valign="top">66.88</td></tr><tr><td align="left" valign="top">mBERT<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">83.42</td><td align="left" valign="top">80.75</td><td align="left" valign="top">73.07</td><td align="left" valign="top">74.32</td><td align="left" valign="top">76.21</td><td align="left" valign="top">77.55</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">80.91</td><td align="left" valign="top">71.39</td><td align="left" valign="top">72.25</td><td align="left" valign="top">70.40</td><td align="left" valign="top">70.82</td><td align="left" valign="top">73.15</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">60.48</td><td align="left" valign="top">52.39</td><td align="left" valign="top">63.28</td><td align="left" valign="top">66.95</td><td align="left" valign="top">62.63</td><td align="left" valign="top">61.15</td><td align="char" char="." valign="top">71.13</td></tr><tr><td align="left" valign="top">mmBERT<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">85.05</td><td align="left" valign="top">83.46</td><td align="left" valign="top">74.75</td><td align="left" valign="top">75.29</td><td align="left" valign="top">74.26</td><td align="left" valign="top">78.56</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">81.46</td><td align="left" valign="top">75.41</td><td align="left" valign="top">73.58</td><td align="left" valign="top">71.03</td><td align="left" valign="top">69.98</td><td align="left" valign="top">74.29</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">66.10</td><td align="left" valign="top">53.49</td><td align="left" valign="top">62.92</td><td align="left" valign="top">64.62</td><td align="left" valign="top">61.32</td><td align="left" valign="top">61.69</td><td align="char" char="." valign="top">63.50</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Target language.</p></fn><fn id="table3fn2"><p><sup>b</sup>Other language.</p></fn><fn id="table3fn3"><p><sup>c</sup>Not included in the average and serves only as an additional reference (its test set size differs from the others).</p></fn><fn id="table3fn4"><p><sup>d</sup>XLMR: XLM-Roberta.</p></fn><fn id="table3fn5"><p><sup>e</sup>mBERT: Multilingual BERT (Bidirectional Encoder Representations from Transformers).</p></fn><fn id="table3fn6"><p><sup>f</sup>mmBERT: Multilingual ModernBERT.</p></fn></table-wrap-foot></table-wrap><p>Detailed results for all training configurations, including those augmented with machine-translated data, are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. This appendix also reports a detailed breakdown of results for possible and impossible instances across all test question instances. This distinction allows us to evaluate not only how accurately the models extract evidences for questions when the corresponding evidence is present in the report (possible instances), but also how reliably they can predict that a report does not contain an answer to a given question (impossible instances).</p><p>We observe that data augmentation of all investigated types (<inline-formula><mml:math id="ieqn4"><mml:mi>O</mml:mi><mml:mover><mml:mrow><mml:mo>&#x2192;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:mover><mml:mi>T</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn5"><mml:mi>O</mml:mi><mml:mover><mml:mrow><mml:mo>&#x2192;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:mover><mml:mi>O</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn6"><mml:mi>T</mml:mi><mml:mover><mml:mrow><mml:mo>&#x2192;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:mover><mml:mi>O</mml:mi></mml:math></inline-formula>) can improve suboptimal training configurations, particularly in scenarios where training data in the target language are scarce or when a single multilingual model is desired. However, augmentation does not benefit the best-performing monolingually fine-tuned models trained in the target language&#x2013;only setting (<italic>T</italic>); on the contrary, it generally leads to a decrease in performance.</p><p>All 3 models achieve near-perfect performance in correctly predicting that no relevant supporting evidence exists in the text for impossible questions. The main challenge lies in questions that do have supporting evidence but where the models fail to locate it. For these instances, the models achieve an average <italic>F</italic><sub>1</sub>-score of over 60% and an EM of around 45%, which remains highly valuable: in nearly 50% of these possible instances, the models are able to identify all supporting evidence exactly. Interestingly, performance varies notably across languages: for Greek and Bulgarian, both <italic>F</italic><sub>1</sub> and EM scores are particularly high, whereas for the remaining 3 languages, performance is comparatively lower.</p><p>In <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>, we further provide a detailed breakdown of performance across different question data types, including Boolean, multiple-choice enumeration, date-time, open-ended string, and number questions. While Boolean, enumeration, and integer questions achieve consistently high scores across languages, date-time, open-ended string, and number questions exhibit substantial variability, highlighting the impact of diverse writing conventions and styles across countries on model performance.</p><p>In <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>, we provide a detailed error analysis of the evidence extraction task using the mmBERT model.</p></sec><sec id="s3-2"><title>Answer Prediction (S2)</title><p>Similar to the previous component of our pipeline, evidence extraction (S1), we investigate whether Answer Prediction (S2) can rely on a single shared model trained jointly on all original training reports across all languages (<italic>T+O</italic>), or whether multilingual training introduces interference and it is preferable to train a separate model for each language using only reports written in the target language (<italic>T</italic>). We further investigate what happens when no training reports are available in the target language and the model has to rely solely on reports from the remaining languages (<italic>O</italic>). In addition, we examine whether model size (4B vs 8B parameters) plays a role and whether clinical pretraining is an important factor. <xref ref-type="table" rid="table4">Table 4</xref> reports accuracy for different training configurations and models. We observe that, similarly to S1, the best results are obtained in the monolingual setting (<italic>T</italic>), where a separate model is trained for each language. The highest performance is achieved by Med42, reaching an average accuracy of 95%. Nevertheless, the multilingual shared model trained in the <italic>T+O</italic> setting exhibits only a minimal decrease in performance. The absence of reports in the target language has a substantial impact on the results, although this effect is considerably less severe than for the S1 component. Furthermore, model size does not appear to play a major role, as all evaluated models achieve very similar performance. Likewise, clinical pretraining does not appear to provide a particularly strong advantage.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Answer Prediction (S2) results for models trained in various training settings.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">T<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="bottom">O<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="bottom">Bulgarian</td><td align="left" valign="bottom">Greek</td><td align="left" valign="bottom">English</td><td align="left" valign="bottom">Polish</td><td align="left" valign="bottom">Romanian</td><td align="left" valign="bottom">Average</td><td align="left" valign="bottom">Spanish<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">MediPhi</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">92.79</td><td align="left" valign="top">97.03</td><td align="left" valign="top">93.54</td><td align="left" valign="top">92.02</td><td align="left" valign="top">96.12</td><td align="left" valign="top">94.30</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="char" char="." valign="top">92.90</td><td align="char" char="." valign="top">97.21</td><td align="char" char="." valign="top">93.45</td><td align="char" char="." valign="top">91.47</td><td align="char" char="." valign="top">96.83</td><td align="char" char="." valign="top">94.37</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">85.03</td><td align="left" valign="top">91.91</td><td align="left" valign="top">85.38</td><td align="left" valign="top">82.77</td><td align="left" valign="top">92.39</td><td align="left" valign="top">87.49</td><td align="char" char="." valign="top">85.26</td></tr><tr><td align="left" valign="top" colspan="3">Phi3.5 Mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">92.81</td><td align="left" valign="top">97.25</td><td align="left" valign="top">93.85</td><td align="left" valign="top">91.77</td><td align="left" valign="top">96.40</td><td align="left" valign="top">94.41</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">92.88</td><td align="left" valign="top">97.39</td><td align="left" valign="top">92.97</td><td align="left" valign="top">91.58</td><td align="left" valign="top">96.76</td><td align="left" valign="top">94.31</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">84.48</td><td align="left" valign="top">94.23</td><td align="left" valign="top">83.85</td><td align="left" valign="top">81.41</td><td align="left" valign="top">92.75</td><td align="left" valign="top">87.34</td><td align="char" char="." valign="top">85.90</td></tr><tr><td align="left" valign="top" colspan="3">MedGemma</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">92.54</td><td align="left" valign="top">97.48</td><td align="left" valign="top">93.56</td><td align="left" valign="top">92.33</td><td align="left" valign="top">96.86</td><td align="left" valign="top">94.55</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">93.14</td><td align="left" valign="top">97.65</td><td align="left" valign="top">93.52</td><td align="left" valign="top">91.83</td><td align="left" valign="top">96.91</td><td align="left" valign="top">94.61</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">86.56</td><td align="left" valign="top">95.32</td><td align="left" valign="top">87.72</td><td align="left" valign="top">85.42</td><td align="left" valign="top">91.96</td><td align="left" valign="top">89.40</td><td align="char" char="." valign="top">86.57</td></tr><tr><td align="left" valign="top" colspan="3">Gemma3</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">92.55</td><td align="left" valign="top">98.01</td><td align="left" valign="top">93.72</td><td align="left" valign="top">91.87</td><td align="left" valign="top">96.61</td><td align="left" valign="top">94.55</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">92.42</td><td align="left" valign="top">97.65</td><td align="left" valign="top">93.14</td><td align="left" valign="top">91.36</td><td align="left" valign="top">96.80</td><td align="left" valign="top">94.27</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">86.23</td><td align="left" valign="top">95.33</td><td align="left" valign="top">87.14</td><td align="left" valign="top">84.47</td><td align="left" valign="top">91.77</td><td align="left" valign="top">88.99</td><td align="char" char="." valign="top">86.51</td></tr><tr><td align="left" valign="top" colspan="3">LLaMA3</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">92.95</td><td align="left" valign="top">97.79</td><td align="left" valign="top">94.49</td><td align="left" valign="top">92.71</td><td align="left" valign="top">97.03</td><td align="left" valign="top">94.99</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">93.18</td><td align="left" valign="top">97.56</td><td align="left" valign="top">94.31</td><td align="left" valign="top">91.73</td><td align="left" valign="top">96.95</td><td align="left" valign="top">94.75</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">85.36</td><td align="left" valign="top">95.74</td><td align="left" valign="top">87.21</td><td align="left" valign="top">82.57</td><td align="left" valign="top">92.39</td><td align="left" valign="top">88.65</td><td align="char" char="." valign="top">84.14</td></tr><tr><td align="left" valign="top" colspan="3">BioMistral</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">93.09</td><td align="left" valign="top">97.68</td><td align="left" valign="top">94.80</td><td align="left" valign="top">92.42</td><td align="left" valign="top">97.13</td><td align="left" valign="top">95.02</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">93.24</td><td align="left" valign="top">97.82</td><td align="left" valign="top">93.30</td><td align="left" valign="top">92.27</td><td align="left" valign="top">97.16</td><td align="left" valign="top">94.76</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">86.37</td><td align="left" valign="top">92.67</td><td align="left" valign="top">86.55</td><td align="left" valign="top">85.67</td><td align="left" valign="top">93.74</td><td align="left" valign="top">89.00</td><td align="char" char="." valign="top">86.85</td></tr><tr><td align="left" valign="top" colspan="3">Mistral</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">93.27</td><td align="left" valign="top">97.76</td><td align="left" valign="top">94.57</td><td align="left" valign="top">92.60</td><td align="left" valign="top">97.16</td><td align="left" valign="top">95.07</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">93.21</td><td align="left" valign="top">97.74</td><td align="left" valign="top">93.34</td><td align="left" valign="top">92.26</td><td align="left" valign="top">97.00</td><td align="left" valign="top">94.71</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">85.14</td><td align="left" valign="top">93.12</td><td align="left" valign="top">86.57</td><td align="left" valign="top">84.45</td><td align="left" valign="top">93.51</td><td align="left" valign="top">88.56</td><td align="char" char="." valign="top">86.26</td></tr><tr><td align="left" valign="top" colspan="3">Med42</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">93.04</td><td align="left" valign="top">97.82</td><td align="left" valign="top">94.88</td><td align="left" valign="top">92.75</td><td align="left" valign="top">96.91</td><td align="left" valign="top">95.08</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">93.06</td><td align="left" valign="top">97.84</td><td align="left" valign="top">94.04</td><td align="left" valign="top">92.06</td><td align="left" valign="top">97.16</td><td align="left" valign="top">94.83</td><td align="char" char="." valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x00D7;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">86.45</td><td align="left" valign="top">95.36</td><td align="left" valign="top">87.27</td><td align="left" valign="top">84.73</td><td align="left" valign="top">93.38</td><td align="left" valign="top">89.44</td><td align="char" char="." valign="top">86.40</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Target language.</p></fn><fn id="table4fn2"><p><sup>b</sup>Other language.</p></fn><fn id="table4fn3"><p><sup>c</sup>Not included in the average and serves only as an additional reference (its test set size differs from the others).</p></fn></table-wrap-foot></table-wrap><p><xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> provides a detailed breakdown of results obtained using the <inline-formula><mml:math id="ieqn7"><mml:mi>T</mml:mi><mml:mover><mml:mrow><mml:mo>&#x2192;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:mover><mml:mi>O</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn8"><mml:mi>O</mml:mi><mml:mover><mml:mrow><mml:mo>&#x2192;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:mover><mml:mi>T</mml:mi></mml:math></inline-formula>, and <inline-formula><mml:math id="ieqn9"><mml:mi>O</mml:mi><mml:mover><mml:mrow><mml:mo>&#x2192;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:mover><mml:mi>O</mml:mi></mml:math></inline-formula> augmentation strategies. It also reports separate results for possible and impossible instances only. Similar to the previous case, augmentation has neither a substantial positive nor a negative effect on performance. However, for impossible questions, Greek and Romanian exhibit highly consistent default predictions, achieving near 100% accuracy, whereas for Bulgarian and Polish the model is more frequently surprised, though still maintains strong accuracy above 90%. But also for possible questions, the models demonstrate strong performance, achieving an average accuracy of approximately 93%.</p><p><xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> provides a detailed breakdown of results according to the answer data type. Boolean questions are the easiest, and enumeration questions also achieve consistently high accuracy across languages. Other question types show considerable variability and inconsistent performance, reflecting the differences in how reports are written in different countries and languages.</p><p><xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref> presents the results of the models without fine-tuning, operating solely in a few-shot setting. When considering the 20-shot setting, the models achieve accuracy of up to 90%, indicating that target-language training improves performance by only 2&#x2010;3 percentage points.</p><p>We already know that Evidence Extraction (S1) can be a bottleneck in the overall pipeline, as we achieve on average only 78% EM with the gold annotations, and it can impose a substantial burden on clinicians in the HITL setting. Identifying or typing the regions of evidence within paragraph-level retrieval settings may be a simpler task [<xref ref-type="bibr" rid="ref59">59</xref>]. However, this increases the challenge for the generative model used for Answer Prediction (S2), which must operate over longer contexts. It also poses additional difficulty for medical experts in the HITL setting, who need to quickly locate and verify the relevant evidences within retrieved extended text segments. To assess how generative models could potentially handle longer text segments, <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref> shows the effect of extended contexts on the Med42 model. We show that substantially longer contexts, up to 25 times longer than the evidence spans themselves, have only a minimal impact on model performance, demonstrating the robustness of generative models and highlighting promising directions for future research.</p></sec><sec id="s3-3"><title>Evidence-Based Question Answering (S1-S2)</title><p>In the previous sections, we evaluated Evidence Extraction (S1) and Answer Prediction (S2) independently. From the Evidence Extraction (S1) experiments, mmBERT was found to perform best in the <italic>T</italic> setting. For Answer Prediction (S2), the Med42 model exhibited the most stable performance. Both models are trained using gold data; specifically, the Med42 model for Answer Prediction (S2) is trained on gold evidence. During evaluation, however, Med42 is provided with predicted evidences generated by the mmBERT model from the Evidence Extraction module rather than the gold evidences.</p><p><xref ref-type="table" rid="table5">Table 5</xref> illustrates a comparison of Answer Prediction (S2) performance using gold evidence from the previous Answer Prediction (S2) section under Results versus Evidence-Based Question-Answering (S1-S2) performance using predicted evidences. The table also includes a baseline based on the prompted LLM gpt-oss-120B [<xref ref-type="bibr" rid="ref82">82</xref>] operating in a zero-shot setting, where no evidence spans are provided or requested, and no human validator is involved at any stage of the process, only the report text and set of questions. Despite an average drop of approximately 8%, the model still achieves 88.47% accuracy for correctly filling in form questions, with 77.19% accuracy on <italic>possible</italic> questions and 94.91% accuracy on <italic>impossible</italic> questions. This confirms that <italic>possible</italic> questions represent the most challenging aspect of the full pipeline, as their performance drops by nearly 20% compared with gold evidences.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Evidence-Based Question-Answering (Evidence Extraction [S1]+Answer Prediction [S2]) performance compared with baselines.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Setting</td><td align="left" valign="bottom">Bulgarian</td><td align="left" valign="bottom">Greek</td><td align="left" valign="bottom">English</td><td align="left" valign="bottom">Polish</td><td align="left" valign="bottom">Romanian</td><td align="left" valign="bottom">Average</td></tr></thead><tbody><tr><td align="left" valign="top">Full testset</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>gpt-oss-120B zero-shot (S1-S2)</td><td align="left" valign="top">77.25</td><td align="left" valign="top">86.58</td><td align="left" valign="top">69.28</td><td align="left" valign="top">79.73</td><td align="left" valign="top">81.99</td><td align="left" valign="top">78.97</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Answer Prediction (S2) - gold evidences</td><td align="left" valign="top">93.04</td><td align="left" valign="top">97.82</td><td align="left" valign="top">94.88</td><td align="left" valign="top">92.75</td><td align="left" valign="top">96.91</td><td align="left" valign="top">95.08</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Evidence-Based Question-Answering (S1-S2)</td><td align="left" valign="top">89.09</td><td align="left" valign="top">93.87</td><td align="left" valign="top">86.66</td><td align="left" valign="top">83.78</td><td align="left" valign="top">88.96</td><td align="left" valign="top">88.47</td></tr><tr><td align="left" valign="top">Possible instances</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>gpt-oss-120B zero-shot (S1-S2)</td><td align="left" valign="top">76.20</td><td align="left" valign="top">83.62</td><td align="left" valign="top">71.15</td><td align="left" valign="top">71.89</td><td align="left" valign="top">73.76</td><td align="left" valign="top">75.32</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Answer Prediction (S2) - gold evidences</td><td align="left" valign="top">94.95</td><td align="left" valign="top">95.45</td><td align="left" valign="top">92.03</td><td align="left" valign="top">91.27</td><td align="left" valign="top">94.78</td><td align="left" valign="top">93.70</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Evidence-Based Question-Answering (S1-S2)</td><td align="left" valign="top">85.26</td><td align="left" valign="top">88.36</td><td align="left" valign="top">72.29</td><td align="left" valign="top">64.11</td><td align="left" valign="top">75.92</td><td align="left" valign="top">77.19</td></tr><tr><td align="left" valign="top">Impossible instances</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>gpt-oss-120B zero-shot (S1-S2)</td><td align="left" valign="top">77.78</td><td align="left" valign="top">89.28</td><td align="left" valign="top">68.19</td><td align="left" valign="top">83.32</td><td align="left" valign="top">86.92</td><td align="left" valign="top">81.10</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Answer Prediction (S2) - gold evidences</td><td align="left" valign="top">92.07</td><td align="left" valign="top">99.97</td><td align="left" valign="top">96.55</td><td align="left" valign="top">93.42</td><td align="left" valign="top">98.18</td><td align="left" valign="top">96.04</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Evidence-Based Question-Answering (S1-S2)</td><td align="left" valign="top">91.04</td><td align="left" valign="top">98.87</td><td align="left" valign="top">95.09</td><td align="left" valign="top">92.78</td><td align="left" valign="top">96.76</td><td align="left" valign="top">94.91</td></tr></tbody></table></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this study, we show that clinical Question Answering over stroke discharge summaries can be effectively framed as an Evidence-Based Question Answering (S1-S2) problem across 5 languages (Bulgarian, Greek, English, Polish, and Romanian). This formulation exposes an intermediate step, Evidence Extraction (S1), in which the model identifies specific text spans as evidences within the discharge summary, enabling fast HITL validation for stroke registry workflows and also allowing the use of RAG-style modeling without requiring the very large computational resources needed to fit or process entire reports directly in an LLM. Even without evidence HITL validation, the end-to-end system achieves 88% accuracy in form filling, including 77% for patient-specific questions (<italic>possible</italic>) and 95% for default or unverifiable items (<italic>impossible</italic>).</p><p>Evidence Extraction (S1), the component responsible for identifying supporting text spans in the report for a given question, proved to be the bottleneck of the overall Evidence-Based Question-Answering (S1-S2) pipeline. Although detecting cases in which no evidence is present for a given question within a report was nearly perfect (ie, for <italic>impossible</italic> questions), extracting the specific supporting spans was substantially more challenging, reaching only 62% <italic>F</italic><sub>1</sub> and 45% EM. But still, as a result, the system was able to surface sufficiently relevant text to assist a human validator in roughly half of the cases. After fine-tuning, we observed on English reports that multilingual general-domain models performed even better than clinically pretrained models, suggesting that it is not clear whether clinical pretraining would provide any additional benefit for this component.</p><p>Answer Prediction (S2), where the model predicts the final form answer based on extracted and human-validated evidences, was the more stable component of the pipeline. Accuracy reached 95% overall (94% for questions with available evidences and 96% for cases with no evidence), and performance remained robust even when provided with evidence windows more than 20 times longer than the concise spans. This indicates that Answer Prediction (S2) does not require highly focused evidence to generate reliable outputs, though whether this tolerance to noisy input translates to human performance in HITL settings remains an open question. Clinical pretraining does not appear beneficial for this component either, as base models and clinically pretrained counterparts behave similarly. Moreover, smaller models achieve nearly comparable performance, with only a minor drop despite having roughly half the parameter count.</p><p>Performance also varied substantially across question types. Boolean and enumeration questions were handled consistently well, whereas date-time, numeric, and open-ended string questions were more error-prone and showed larger discrepancies across languages. These differences correlate with the distribution of question types in the dataset, making it unclear whether the increased difficulty reflects intrinsic task complexity or simply underrepresentation during training.</p></sec><sec id="s4-2"><title>Cross-Language Variation</title><p>The results indicate that observed differences between languages are primarily due to variations in reporting practices rather than intrinsic language aspects. Discharge summaries from different countries and hospitals varied in length, structure, and narrative detail, as reflected in our dataset statistics, which led to heterogeneous difficulty across languages. Furthermore, certain languages tended to use more consistent default values for <italic>impossible</italic> questions, whereas others exhibited more variable or implicit phrasing, which affected model performance. Similarly, differences in question data types led to varying difficulty across languages, with some types being easier to answer in certain languages than in others.</p><p>Multilingual and augmented training had divergent effects across components. For Evidence Extraction (S1), including reports from other languages (or their machine-translation&#x2013;based augmentations) during training consistently reduced performance, indicating that cross-language mixing introduces noise rather than useful generalization. In contrast, augmenting target-language reports into other languages did not cause any issues, confirming that the challenge lies in report content and writing style rather than the language itself. For Answer Prediction (S2), performance was largely insensitive to multilingual training and comparable across model sizes, suggesting that a single shared model is sufficient for this component.</p><p>Generalization to new, unseen languages cannot be guaranteed under current conditions, as both Evidence Extraction (S1) and Answer Prediction (S2) rely on the presence of original target-language reports during training to achieve meaningful performance; without them, accuracy drops substantially.</p><p>To better understand the sources of the observed cross-language performance differences, we conducted an additional analysis of the underlying data and identified the following cross-linguistic differences in reporting practices. Our experimental findings suggest that reporting conventions vary substantially across countries and languages and that these differences affect model performance. While we already compared report length and structural differences across languages, we further analyzed the reports to identify specific discrepancies in reporting practices that may contribute to the observed performance variation.</p><p>We found that reports in Bulgarian use mmol/L as the scale for cholesterol, whereas other languages predominantly use mg/dL. Although the official scale for annotations was mmol/L, only 46% of Polish annotators converted values from mg/dL to mmol/L. In the remaining languages, no conversion was performed. Additionally, Bulgarian reports use the Glasgow&#x2013;Li&#x00E8;ge Coma Scale, while reports from other countries rely on the Glasgow Coma Scale. In English reports, the patient&#x2019;s age is not explicitly stated. Instead, only the date of birth is provided, requiring the age to be derived using the admission date. In contrast, reports in the other languages explicitly provide the patient&#x2019;s age.</p><p>Date and time annotation conventions also vary significantly across languages, particularly with respect to the inclusion of seconds and the balance between strictly numeric formats and narrative temporal expressions. Bulgarian and Polish reports are distinct in their dual approach and frequently combine natural language phrases such as &#x201C;the day before&#x201D; with numeric dates that generally omit seconds. Bulgarian follows a day-first format with dot separators (DD.MM.YYYY HH:mm), whereas Polish adopts the ISO-8601 year-first format with hyphens (YYYY-MM-DD HH:mm). Spanish is the only language in the dataset to explicitly include seconds in its standard notation, allowing both dot- and slash-separated day-first formats (DD.MM.YYYY HH:mm:ss or DD/MM/YYYY HH:mm:ss). The remaining languages rely on day-first numeric timestamps without seconds but differ in formatting details. Romanian favors a 2-digit year (DD.MM.YY HH:mm), Greek shows inconsistency in separators and zero-padding (DD.M.YYYY HH:mm or DD/MM/YYYY HH:mm), and English reports generally follow the European convention using slashes (DD/MM/YYYY HH:mm).</p></sec><sec id="s4-3"><title>Practical Implications for Clinical Use</title><p>The findings indicate that Evidence-Based Question Answering (S1-S2) can meaningfully support structured data collection from long discharge summaries about stroke patients in multiple languages, especially when paired with HITL validation. Highlighted evidence spans can help speed up verification and support the validation of automatically generated answers against the report. Preliminary internal measurements suggest that the NLP-assisted annotation workflow reduces overall clinician annotation effort by approximately 25% across the full Evidence-Based Question Answering (S1-S2) pipeline.</p><p>Answer Prediction (S2), operating on extracted and human-validated evidence, performs very well across languages. For practical deployment, it is advantageous to directly integrate questions for which sufficient training data exists and for which reliable performance is observed, such as Boolean and enumeration questions.</p><p>From a computational perspective, Evidence Extraction (S1) requires a separate model for each language; however, the encoder models we use for Evidence Extraction (S1) are relatively small (up to 307M parameters) and easily deployable. In contrast, for Answer Prediction (S2), which uses substantially larger models, a single shared model is sufficient for all languages, with 4B parameters providing strong performance and making larger 8B models unnecessary.</p></sec><sec id="s4-4"><title>Limitations</title><p>This study has several limitations. Both stages of our Evidence-Based Question Answering (S1-S2) pipeline rely on transformer models that may exceed available computational capacity in some hospital settings. The evaluation covers 6 languages, with a very limited number of reports available for Spanish and limited training instances for some question types. The Spanish dataset (27 reports) is used only as a small-scale test reference and was not part of a fully powered evaluation; zero-shot transfer to Spanish was therefore only explored rather than systematically evaluated. Findings cannot automatically be extended to languages with substantially different reporting structures or insufficient training data.</p></sec><sec id="s4-5"><title>Future Directions</title><p>Future work may explore retrieval strategies that operate at paragraph or section level to mitigate the sensitivity of span extraction. Lightweight architectures or distillation approaches could improve deployability in resource-constrained environments. Extending the framework to additional clinical domains and languages would help assess broader generalizability. Human-centered evaluations, such as systematic time-motion studies measuring verification time, cognitive load, error correction effort, and clinician trust, would provide direct and rigorous evidence of practical impact in real-world workflows focused on how the system accelerates completion of stroke registries such as RES-Q, where clinicians directly fill standardized form fields without requiring explicit evidence span or other annotation.</p><p>More complex cross-language variation analysis and bias auditing across languages may further be useful in the form of measuring differences in the frequency and distribution of evidence spans and relevant sections at different positions in the text, as well as how reporting conventions vary in detail not only between languages but also between reports within the same language across different hospitals and settings.</p></sec><sec id="s4-6"><title>Comparison With Previous Work</title><p>Although earlier studies did not directly address Evidence-Based Question Answering (S1-S2) in the same sense as we do, various works have focused on clinical information extraction from discharge summaries. For this purpose, the n2c2 shared task [<xref ref-type="bibr" rid="ref83">83</xref>] was introduced, targeting named entity recognition, concept extraction, relation classification, and end-to-end information extraction systems across different discharge summaries. Subsequent research explored recurrent neural network-based approaches [<xref ref-type="bibr" rid="ref84">84</xref>], which were outperformed by encoder-based transformer models, such as BERT [<xref ref-type="bibr" rid="ref85">85</xref>]. The data from this shared task also served as the basis for creating the emrQA dataset [<xref ref-type="bibr" rid="ref58">58</xref>], which contains synthetic questions aimed at locating explicit evidence spans in text, similar to our Evidence Extraction (S1) task. In this setting as well, encoder-based transformer models achieve the strongest performance [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref60">60</xref>].</p><p>Beyond emrQA, several recent datasets derived from the MIMIC corpus [<xref ref-type="bibr" rid="ref86">86</xref>] further extend discharge-summary-based Question-Answering tasks in directions more closely aligned with our Answer Prediction (S2). Unlike our approach, which targets short, schema-constrained answers, these resources focus on generating and evaluating richer, free-text responses in natural clinical language. EHRNoteQA [<xref ref-type="bibr" rid="ref87">87</xref>] provides a benchmark for evaluating language models on complex clinical questions that may require reasoning across multiple discharge summaries. Its automatically generated question-answer pairs were refined by clinicians, and the dataset is used to assess a broad range of decoder-style transformer models in both open-ended and multiple-choice formats. MeDiSumQA [<xref ref-type="bibr" rid="ref88">88</xref>] similarly constructs a dataset from MIMIC discharge summaries with a focus on patient-oriented question answering; its automatically generated question-answer pairs are likewise refined through manual quality control and are used to benchmark both general-purpose and biomedical language models on producing layperson-friendly responses. EHR-DS-QA [<xref ref-type="bibr" rid="ref89">89</xref>], by contrast, adopts a fully synthetic pipeline in which question-answer pairs are generated directly from individual discharge summaries to support the development of retrieval-augmented language models for clinical information extraction.</p><p>On MIMIC [<xref ref-type="bibr" rid="ref90">90</xref>], ArchEHR-QA shared task [<xref ref-type="bibr" rid="ref91">91</xref>] has also been established, representing a different variant of a combination of Evidence Extraction (S1) and Answer Prediction (S2) tasks. The goal is both to predict answers to patient questions about their discharge summaries and to ensure that all answer content is explicitly grounded in the source discharge summary sentences (which allows fast validation). For simplicity, the task provides shortened discharge summaries ranging from tens to a few dozen sentences. While using third-party LLMs for end-to-end prompting might seem attractive, the underlying data are sensitive MIMIC records and cannot be shared externally. As a result, current methods rely on encoder-based models, such as BERT, to identify relevant sentences, followed by generative LLMs to produce the final patient-facing answers [<xref ref-type="bibr" rid="ref92">92</xref>,<xref ref-type="bibr" rid="ref93">93</xref>]. But the task still remains not fully solved: sentence-level evidence retrieval achieves <italic>F</italic><sub>1</sub>-scores of only 50%-60%, and BLEU scores relative to clinician-authored answers reach roughly 4 to 5.</p><p>In the stroke clinical domain, previous work has examined how NLP can be applied to predictive modeling tasks [<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref40">40</xref>]. These studies consider both classification and regression settings and use machine learning classifiers and regression models. They consistently show that incorporating unstructured clinical text from EHRs (such as discharge-related notes, histories of present illness, or imaging reports) improves predictive performance compared with models relying solely on structured data. Subsequent work focuses on extracting a broader set of stroke-related information from unstructured clinical text [<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref44">44</xref>]. These approaches implement rule-based methods, classical machine learning models, and recurrent neural networks. For well-defined, relatively unambiguous tasks such as detecting large-vessel occlusion or silent brain infarcts, reported metrics (accuracy or area under the curve) are often high, around 95% or more. More complex or implicitly expressed attributes, such as ischemia extent, ASPECTS scores, or collateral status, typically show substantially lower performance.</p><p>More recent studies have explored the use of LLMs for stroke-related information extraction directly from unstructured clinical text [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref49">49</xref>]. These works formulate document-level classification tasks and schema-defined field extraction over clinical notes and discharge summaries, using prompt-based LLM inference without task-specific fine-tuning. In particular, a third-party LLM achieves 98% accuracy for coarse-grained stroke type classification, while sensitivity for ischemic stroke subtypes varies substantially, ranging from 40% to 95% depending on the subtype. Complementarily, a locally deployed LLM is shown to extract heterogeneous stroke audit variables from free-text discharge summaries with an overall item-level accuracy of 94%.</p><p>NLP methods have also proven useful in other clinical domains (eg, oncology), where BERT-based models achieved 98% accuracy for histology extraction [<xref ref-type="bibr" rid="ref46">46</xref>] or 99.9% average accuracy for breast cancer pathology extraction, substantially outperforming rule-based methods [<xref ref-type="bibr" rid="ref47">47</xref>], or gynecology, where only rule-based approaches were tested, reaching 83% <italic>F</italic><sub>1</sub>-score for surgical history extraction [<xref ref-type="bibr" rid="ref45">45</xref>].</p><p>All of the studies mentioned so far were conducted and evaluated primarily in English. However, some works have explored multilinguality in the clinical domain. Automatically expanding training data to additional languages does not necessarily yield substantial gains [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref94">94</xref>], and clinical pretraining of language models may not always be crucial [<xref ref-type="bibr" rid="ref31">31</xref>]. Nevertheless, several studies specifically target information extraction from original reports in other languages, including German and French [<xref ref-type="bibr" rid="ref94">94</xref>], Italian [<xref ref-type="bibr" rid="ref95">95</xref>], Portuguese [<xref ref-type="bibr" rid="ref96">96</xref>], or Dutch [<xref ref-type="bibr" rid="ref97">97</xref>].</p><p>Our work builds on these prior NLP-based clinical information extraction studies, leveraging modern methods combining BERT-like models with LLMs. In contrast to previous work, we reveal challenges related to multilinguality, diverse reporting conventions across hospitals and countries, and varying question types within long stroke discharge summaries, providing the first cross-lingual comparison in this domain. We present the first results on a novel multilingual evidence-based question-answering dataset, achieving 89% end-to-end accuracy (77% for patient-specific questions and 95% for default items), and propose a novel task framing that decomposes the problem into evidence extraction (S1) and answer prediction (S2) to enable rapid HITL validation.</p></sec><sec id="s4-7"><title>Conclusions</title><p>This study shows that clinical Question Answering over multilingual stroke discharge summaries can be framed as an Evidence-Based Question-Answering task, with an intermediate Evidence Extraction step enabling HITL validation and allowing effective use of LLMs without relying on large computational resources and full-report context processing. While Evidence Extraction remains the main bottleneck, Answer Prediction is notably robust across languages and model sizes. Our findings indicate that the approach can meaningfully support structured data collection, particularly for well-represented question types, and can be deployed without excessive model requirements. However, generalization to new languages remains constrained by the need for target-language training data. Future work should evaluate the framework in additional clinical settings and assess its practical impact on real-world workflows.</p></sec></sec></body><back><ack><p>We thank the clinicians from the RES-Q+ project for their extensive annotation work and acknowledge the RES-Q+ software engineering team for developing and maintaining the annotation tool that enabled the data collection process. We also used the generative AI tools, including GPT-4o (OpenAI) and Gemini 1.5 (Google), for code optimization, proofreading and editing, and reformatting, with all outputs fully reviewed and verified by the authors.</p></ack><notes><sec><title>Funding</title><p>This research received support and funding from the European Union&#x2019;s Horizon Europe research and innovation programme project RES-Q plus (grant agreement 101057603). Views and opinions expressed are, however, those of the authors only and do not necessarily reflect those of the European Union or the Health and Digital Executive Agency. This work was also partially supported by the Charles University GAUK grant 284125.</p></sec><sec><title>Data Availability</title><p>The dataset used in this study contains sensitive clinical information and cannot be shared outside the participating institutions due to contractual and patient confidentiality constraints.</p><p>The code developed for this project is provided in <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>.</p></sec></notes><fn-group><fn fn-type="con"><p>VL conceptualized the study, designed its structure, conducted all experiments, and wrote the original draft. AD performed additional comparative experiments, analyzed data and annotations, and investigated cross-lingual conventions in discharge reports. SB supervised AD and contributed to analysis and interpretation. JM contributed to data format analysis and annotation visualization. SB, JM, and &#x0160;Z consulted with clinicians on specific issues and requirements. RM contributed to clinical framing and direction of the study, including grounding it in real-world clinical use scenarios, and to interpretation and manuscript revision. PP supervised the work and contributed to consultations and manuscript revision. The final version of the paper has been reviewed and approved by all authors.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb2">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb3">EM</term><def><p>exact match</p></def></def-item><def-item><term id="abb4">HITL</term><def><p>human in the loop</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">mBERT</term><def><p>Multilingual BERT</p></def></def-item><def-item><term id="abb7">mmBERT</term><def><p>Multilingual ModernBERT</p></def></def-item><def-item><term id="abb8">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb9">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb10">RES-Q</term><def><p>Registry of Stroke Care Quality</p></def></def-item><def-item><term id="abb11">XLMR</term><def><p>XLM-Roberta</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ivers</surname><given-names>N</given-names> </name><name name-style="western"><surname>Jamtvedt</surname><given-names>G</given-names> </name><name name-style="western"><surname>Flottorp</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Audit and feedback: effects on professional practice and healthcare outcomes</article-title><source>Cochrane Database of Syst Rev</source><year>2012</year><volume>2012</volume><issue>7</issue><fpage>CD000259</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD000259.pub3</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harrison</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hinchcliff</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Manias</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Can feedback approaches reduce unwarranted clinical variation? A systematic rapid evidence synthesis</article-title><source>BMC Health Serv Res</source><year>2020</year><month>01</month><day>16</day><volume>20</volume><issue>1</issue><fpage>40</fpage><pub-id pub-id-type="doi">10.1186/s12913-019-4860-0</pub-id><pub-id pub-id-type="medline">31948447</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ali</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Visser</surname><given-names>EH</given-names> </name><name name-style="western"><surname>West</surname><given-names>RL</given-names> </name><name name-style="western"><surname>van Noord</surname><given-names>D</given-names> </name><name name-style="western"><surname>van der Woude</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>van Deen</surname><given-names>WK</given-names> </name></person-group><article-title>Reporting feedback on healthcare outcomes to improve quality in care: a scoping review</article-title><source>Implement Sci</source><year>2025</year><month>03</month><day>25</day><volume>20</volume><issue>1</issue><fpage>14</fpage><pub-id pub-id-type="doi">10.1186/s13012-025-01424-9</pub-id><pub-id pub-id-type="medline">40133946</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lutz</surname><given-names>N</given-names> </name><name name-style="western"><surname>Alice</surname><given-names>F</given-names> </name><name name-style="western"><surname>Marko</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Data accuracy in the European Cystic Fibrosis Society Patient Registry: results of an on-site data validation project</article-title><source>Orphanet J Rare Dis</source><year>2025</year><month>12</month><day>2</day><volume>20</volume><issue>1</issue><fpage>622</fpage><pub-id pub-id-type="doi">10.1186/s13023-025-04153-w</pub-id><pub-id pub-id-type="medline">41327420</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Doppalapudi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>R</given-names> </name></person-group><article-title>Transforming unstructured digital clinical notes for improved health literacy</article-title><source>Digital Transformation and Society</source><year>2022</year><month>08</month><day>22</day><volume>1</volume><issue>1</issue><fpage>9</fpage><lpage>28</lpage><pub-id pub-id-type="doi">10.1108/DTS-05-2022-0013</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Negro-Calduch</surname><given-names>E</given-names> </name><name name-style="western"><surname>Azzopardi-Muscat</surname><given-names>N</given-names> </name><name name-style="western"><surname>Krishnamurthy</surname><given-names>RS</given-names> </name><name name-style="western"><surname>Novillo-Ortiz</surname><given-names>D</given-names> </name></person-group><article-title>Technological progress in electronic health record system optimization: systematic review of systematic literature reviews</article-title><source>Int J Med Inform</source><year>2021</year><month>08</month><volume>152</volume><fpage>104507</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2021.104507</pub-id><pub-id pub-id-type="medline">34049051</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dinescu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fernandez</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ross</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Karani</surname><given-names>R</given-names> </name></person-group><article-title>Audit and feedback: an intervention to improve discharge summary completion</article-title><source>J Hosp Med</source><year>2011</year><month>01</month><volume>6</volume><issue>1</issue><fpage>28</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.1002/jhm.831</pub-id><pub-id pub-id-type="medline">21241038</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hoque</surname><given-names>DME</given-names> </name><name name-style="western"><surname>Kumari</surname><given-names>V</given-names> </name><name name-style="western"><surname>Hoque</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ruseckaite</surname><given-names>R</given-names> </name><name name-style="western"><surname>Romero</surname><given-names>L</given-names> </name><name name-style="western"><surname>Evans</surname><given-names>SM</given-names> </name></person-group><article-title>Impact of clinical registries on quality of patient care and clinical outcomes: a systematic review</article-title><source>PLoS ONE</source><year>2017</year><volume>12</volume><issue>9</issue><fpage>e0183667</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0183667</pub-id><pub-id pub-id-type="medline">28886607</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Spencer</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Spencer</surname><given-names>SEF</given-names> </name><name name-style="western"><surname>Rodgers</surname><given-names>S</given-names> </name><name name-style="western"><surname>Campbell</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Avery</surname><given-names>AJ</given-names> </name></person-group><article-title>Processing of discharge summaries in general practice: a retrospective record review</article-title><source>Br J Gen Pract</source><year>2018</year><month>08</month><volume>68</volume><issue>673</issue><fpage>e576</fpage><lpage>e585</lpage><pub-id pub-id-type="doi">10.3399/bjgp18X697877</pub-id><pub-id pub-id-type="medline">29914879</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Attipoe</surname><given-names>S</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Schweikhart</surname><given-names>S</given-names> </name><name name-style="western"><surname>Rust</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hoffman</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>S</given-names> </name></person-group><article-title>Factors associated with electronic health record usage among primary care physicians after hours: retrospective cohort study</article-title><source>JMIR Hum Factors</source><year>2019</year><month>09</month><day>30</day><volume>6</volume><issue>3</issue><fpage>e13779</fpage><pub-id pub-id-type="doi">10.2196/13779</pub-id><pub-id pub-id-type="medline">31573912</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Adler-Milstein</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>W</given-names> </name><name name-style="western"><surname>Willard-Grace</surname><given-names>R</given-names> </name><name name-style="western"><surname>Knox</surname><given-names>M</given-names> </name><name name-style="western"><surname>Grumbach</surname><given-names>K</given-names> </name></person-group><article-title>Electronic health records and burnout: time spent on the electronic health record after hours and message volume associated with exhaustion but not with cynicism among primary care clinicians</article-title><source>J Am Med Inform Assoc</source><year>2020</year><month>04</month><day>1</day><volume>27</volume><issue>4</issue><fpage>531</fpage><lpage>538</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocz220</pub-id><pub-id pub-id-type="medline">32016375</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>JX</given-names> </name><name name-style="western"><surname>Goryakin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Maeda</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bruckner</surname><given-names>T</given-names> </name><name name-style="western"><surname>Scheffler</surname><given-names>R</given-names> </name></person-group><article-title>Global health workforce labor market projections for 2030</article-title><source>Hum Resour Health</source><year>2017</year><month>02</month><day>3</day><volume>15</volume><issue>1</issue><fpage>11</fpage><pub-id pub-id-type="doi">10.1186/s12960-017-0187-2</pub-id><pub-id pub-id-type="medline">28159017</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aluttis</surname><given-names>C</given-names> </name><name name-style="western"><surname>Bishaw</surname><given-names>T</given-names> </name><name name-style="western"><surname>Frank</surname><given-names>MW</given-names> </name></person-group><article-title>The workforce for health in a globalized context--global shortages and international migration</article-title><source>Glob Health Action</source><year>2014</year><volume>7</volume><fpage>23611</fpage><pub-id pub-id-type="doi">10.3402/gha.v7.23611</pub-id><pub-id pub-id-type="medline">24560265</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Melton</surname><given-names>GB</given-names> </name><name name-style="western"><surname>Hripcsak</surname><given-names>G</given-names> </name></person-group><article-title>Automated detection of adverse events using natural language processing of discharge summaries</article-title><source>J Am Med Inform Assoc</source><year>2005</year><volume>12</volume><issue>4</issue><fpage>448</fpage><lpage>457</lpage><pub-id pub-id-type="doi">10.1197/jamia.M1794</pub-id><pub-id pub-id-type="medline">15802475</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Doan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bastarache</surname><given-names>L</given-names> </name><name name-style="western"><surname>Klimkowski</surname><given-names>S</given-names> </name><name name-style="western"><surname>Denny</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name></person-group><article-title>Integrating existing natural language processing tools for medication extraction from discharge summaries</article-title><source>J Am Med Inform Assoc</source><year>2010</year><volume>17</volume><issue>5</issue><fpage>528</fpage><lpage>531</lpage><pub-id pub-id-type="doi">10.1136/jamia.2010.003855</pub-id><pub-id pub-id-type="medline">20819857</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Durango</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Torres-Silva</surname><given-names>EA</given-names> </name><name name-style="western"><surname>Orozco-Duque</surname><given-names>A</given-names> </name></person-group><article-title>Named entity recognition in electronic health records: a methodological review</article-title><source>Healthc Inform Res</source><year>2023</year><month>10</month><volume>29</volume><issue>4</issue><fpage>286</fpage><lpage>300</lpage><pub-id pub-id-type="doi">10.4258/hir.2023.29.4.286</pub-id><pub-id pub-id-type="medline">37964451</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meystre</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>GK</given-names> </name><name name-style="western"><surname>Kipper-Schuler</surname><given-names>KC</given-names> </name><name name-style="western"><surname>Hurdle</surname><given-names>JF</given-names> </name></person-group><article-title>Extracting information from textual documents in the electronic health record: a review of recent research</article-title><source>Yearb Med Inform</source><year>2008</year><volume>PMID</volume><issue>1</issue><fpage>128</fpage><lpage>144</lpage><pub-id pub-id-type="doi">10.1055/s-0038-1638592</pub-id><pub-id pub-id-type="medline">18660887</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elvas</surname><given-names>LB</given-names> </name><name name-style="western"><surname>Almeida</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ferreira</surname><given-names>JC</given-names> </name></person-group><article-title>Natural language processing in medical text processing: a scoping literature review</article-title><source>Int J Med Inform</source><year>2025</year><month>12</month><volume>204</volume><fpage>106049</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.106049</pub-id><pub-id pub-id-type="medline">40706199</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Glonti</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hawkesworth</surname><given-names>S</given-names> </name><name name-style="western"><surname>Doupi</surname><given-names>P</given-names> </name><etal/></person-group><article-title>An exploratory analysis of hospital discharge summaries across Europe</article-title><source>Eur J Public Health</source><year>2013</year><month>10</month><day>1</day><volume>23</volume><issue>suppl_1</issue><pub-id pub-id-type="doi">10.1093/eurpub/ckt126.044</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Frings</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rust</surname><given-names>P</given-names> </name><name name-style="western"><surname>Meister</surname><given-names>S</given-names> </name><name name-style="western"><surname>Prinz</surname><given-names>C</given-names> </name><name name-style="western"><surname>Fehring</surname><given-names>L</given-names> </name></person-group><article-title>Diagnosis documentation done right: cross-specialty standard for the diagnosis section in German discharge summaries - a mixed-methods study</article-title><source>J GEN INTERN MED</source><year>2025</year><month>05</month><volume>40</volume><issue>6</issue><fpage>1387</fpage><lpage>1402</lpage><pub-id pub-id-type="doi">10.1007/s11606-025-09395-9</pub-id><pub-id pub-id-type="medline">39915342</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Silver</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Goodman</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Chadha</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Optimizing discharge summaries: a multispecialty, multicenter survey of primary care clinicians</article-title><source>J Patient Saf</source><year>2022</year><month>01</month><day>1</day><volume>18</volume><issue>1</issue><fpage>58</fpage><lpage>63</lpage><pub-id pub-id-type="doi">10.1097/PTS.0000000000000809</pub-id><pub-id pub-id-type="medline">33395016</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>D&#x00F6;ring</surname><given-names>N</given-names> </name><name name-style="western"><surname>Doupi</surname><given-names>P</given-names> </name><name name-style="western"><surname>Glonti</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Electronic discharge summaries in cross-border care in the European Union: how close are we to making it happen?</article-title><source>Int J Care Coord</source><year>2014</year><month>06</month><volume>17</volume><issue>1-2</issue><fpage>38</fpage><lpage>51</lpage><pub-id pub-id-type="doi">10.1177/2053435414540614</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Angelelli</surname><given-names>CV</given-names> </name></person-group><article-title>Cross-border healthcare for all EU residents? Linguistic access in the European Union</article-title><source>Journal of Applied Linguistics and Professional Practice</source><year>2014</year><month>10</month><day>11</day><volume>11</volume><issue>2</issue><fpage>113</fpage><lpage>134</lpage><pub-id pub-id-type="doi">10.1558/japl.31818</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>N&#x00E9;v&#x00E9;ol</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dalianis</surname><given-names>H</given-names> </name><name name-style="western"><surname>Velupillai</surname><given-names>S</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>G</given-names> </name><name name-style="western"><surname>Zweigenbaum</surname><given-names>P</given-names> </name></person-group><article-title>Clinical natural language processing in languages other than English: opportunities and challenges</article-title><source>J Biomed Semantics</source><year>2018</year><month>03</month><day>30</day><volume>9</volume><issue>1</issue><fpage>12</fpage><pub-id pub-id-type="doi">10.1186/s13326-018-0179-8</pub-id><pub-id pub-id-type="medline">29602312</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bengio</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ducharme</surname><given-names>R</given-names> </name><name name-style="western"><surname>Vincent</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jauvin</surname><given-names>C</given-names> </name></person-group><article-title>A neural probabilistic language model</article-title><source>J Mach Learn Res</source><year>2003</year><access-date>2026-07-15</access-date><volume>3</volume><fpage>1137</fpage><lpage>1155</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.jmlr.org/papers/volume3/bengio03a/bengio03a.pdf">https://www.jmlr.org/papers/volume3/bengio03a/bengio03a.pdf</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Devlin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>MW</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>K</given-names> </name><name name-style="western"><surname>Toutanova</surname><given-names>K</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Burstein</surname><given-names>J</given-names> </name><name name-style="western"><surname>Doran</surname><given-names>C</given-names> </name><name name-style="western"><surname>Solorio</surname><given-names>T</given-names> </name></person-group><article-title>BERT: pre-training of deep bidirectional transformers for language understanding</article-title><conf-name>Proceedings of the 2019 Conference of the North</conf-name><conf-date>Jun 2-7, 2019</conf-date><conf-loc>Minneapolis, Minnesota</conf-loc><fpage>4171</fpage><lpage>4186</lpage><pub-id pub-id-type="doi">10.18653/v1/N19-1423</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Grattafiori</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dubey</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jauhri</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The Llama 3 herd of models</article-title><source>arXiv</source><comment>Preprint posted online on  Jul, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2407.21783</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Alsentzer</surname><given-names>E</given-names> </name><name name-style="western"><surname>Murphy</surname><given-names>J</given-names> </name><name name-style="western"><surname>Boag</surname><given-names>W</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Rumshisky</surname><given-names>A</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name></person-group><article-title>Publicly available clinical BERT embeddings</article-title><conf-name>Proc ClinNLP Workshop Association for Computational Linguistics</conf-name><conf-date>Jun 7, 2019</conf-date><conf-loc>Minneapolis, Minnesota, USA</conf-loc><fpage>72</fpage><lpage>78</lpage><pub-id pub-id-type="doi">10.18653/v1/W19-1909</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chiang</surname><given-names>JN</given-names> </name></person-group><article-title>Clinical ModernBERT: an efficient and long context encoder for biomedical text</article-title><source>arXiv</source><comment>Preprint posted online on  Apr, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.03964</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Christophe</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kanithi</surname><given-names>PK</given-names> </name><name name-style="western"><surname>Raha</surname><given-names>T</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Pimentel</surname><given-names>MA</given-names> </name></person-group><article-title>Med42-v2: a suite of clinical LLMs</article-title><source>Arxiv</source><comment>Preprint posted online on  Aug, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2408.06142</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lanz</surname><given-names>V</given-names> </name><name name-style="western"><surname>Pecina</surname><given-names>P</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Demner-Fushman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>D</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>P</given-names> </name></person-group><article-title>When multilingual models compete with monolingual domain-specific models in clinical question answering</article-title><conf-name>Proceedings of the Second Workshop on Patient-Oriented Language Processing (CL4Health)</conf-name><conf-date>May 4, 2025</conf-date><conf-loc>Albuquerque, New Mexico</conf-loc><fpage>69</fpage><lpage>82</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.cl4health-1.6</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Allen</surname><given-names>A</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Zalta</surname><given-names>EN</given-names> </name></person-group><article-title>Privacy and medicine</article-title><source>The Stanford Encyclopedia of Philosophy</source><year>2021</year><access-date>2026-07-15</access-date><publisher-name>Spring</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://plato.stanford.edu/archives/spr2021/entries/privacy-medicine/">https://plato.stanford.edu/archives/spr2021/entries/privacy-medicine/</ext-link></comment></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bani Issa</surname><given-names>W</given-names> </name><name name-style="western"><surname>Al Akour</surname><given-names>I</given-names> </name><name name-style="western"><surname>Ibrahim</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Privacy, confidentiality, security and patient safety concerns about electronic health records</article-title><source>Int Nurs Rev</source><year>2020</year><month>06</month><volume>67</volume><issue>2</issue><fpage>218</fpage><lpage>230</lpage><pub-id pub-id-type="doi">10.1111/inr.12585</pub-id><pub-id pub-id-type="medline">32314398</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="web"><source>ChatGPT</source><year>2025</year><access-date>2026-07-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://chat.openai.com/">https://chat.openai.com/</ext-link></comment></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>AS</given-names> </name></person-group><article-title>Hospital artificial intelligence/machine learning adoption by neighborhood deprivation</article-title><source>Med Care</source><year>2025</year><month>03</month><day>1</day><volume>63</volume><issue>3</issue><fpage>227</fpage><lpage>233</lpage><pub-id pub-id-type="doi">10.1097/MLR.0000000000002110</pub-id><pub-id pub-id-type="medline">39947693</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tian</surname><given-names>A</given-names> </name><etal/></person-group><article-title>A survey on large language model acceleration based on KV cache management</article-title><source>arXiv</source><comment>Preprint posted online on  Dec, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.19442</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sung</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Hsieh</surname><given-names>CY</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>YH</given-names> </name></person-group><article-title>Early prediction of functional outcomes after acute ischemic stroke using unstructured clinical text: retrospective cohort study</article-title><source>JMIR Med Inform</source><year>2022</year><month>02</month><day>17</day><volume>10</volume><issue>2</issue><fpage>e29806</fpage><pub-id pub-id-type="doi">10.2196/29806</pub-id><pub-id pub-id-type="medline">35175201</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sung</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>CH</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>RC</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Jeng</surname><given-names>JS</given-names> </name></person-group><article-title>Natural language processing enhances prediction of functional outcome after acute ischemic stroke</article-title><source>J Am Heart Assoc</source><year>2021</year><month>12</month><day>21</day><volume>10</volume><issue>24</issue><fpage>e023486</fpage><pub-id pub-id-type="doi">10.1161/JAHA.121.023486</pub-id><pub-id pub-id-type="medline">34796719</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lineback</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Garg</surname><given-names>R</given-names> </name><name name-style="western"><surname>Oh</surname><given-names>E</given-names> </name><name name-style="western"><surname>Naidech</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Holl</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Prabhakaran</surname><given-names>S</given-names> </name></person-group><article-title>Prediction of 30-day readmission after stroke using machine learning and natural language processing</article-title><source>Front Neurol</source><year>2021</year><volume>12</volume><fpage>649521</fpage><pub-id pub-id-type="doi">10.3389/fneur.2021.649521</pub-id><pub-id pub-id-type="medline">34326805</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kogan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Twyman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Heap</surname><given-names>J</given-names> </name><name name-style="western"><surname>Milentijevic</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Alberts</surname><given-names>M</given-names> </name></person-group><article-title>Assessing stroke severity using electronic health record data: a machine learning approach</article-title><source>BMC Med Inform Decis Mak</source><year>2020</year><month>01</month><day>8</day><volume>20</volume><issue>1</issue><fpage>8</fpage><pub-id pub-id-type="doi">10.1186/s12911-019-1010-x</pub-id><pub-id pub-id-type="medline">31914991</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>AYX</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>ZA</given-names> </name><name name-style="western"><surname>Pou-Prom</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Automating stroke data extraction from free-text radiology reports using natural language processing: instrument validation study</article-title><source>JMIR Med Inform</source><year>2021</year><month>05</month><day>4</day><volume>9</volume><issue>5</issue><fpage>e24381</fpage><pub-id pub-id-type="doi">10.2196/24381</pub-id><pub-id pub-id-type="medline">33944791</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ong</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Orfanoudaki</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Machine learning and natural language processing methods to identify ischemic stroke, acuity and location from radiology reports</article-title><source>PLoS ONE</source><year>2020</year><volume>15</volume><issue>6</issue><fpage>e0234908</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0234908</pub-id><pub-id pub-id-type="medline">32559211</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Leung</surname><given-names>LY</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Natural language processing for the identification of silent brain infarcts from neuroimaging reports</article-title><source>JMIR Med Inform</source><year>2019</year><month>04</month><day>21</day><volume>7</volume><issue>2</issue><fpage>e12109</fpage><pub-id pub-id-type="doi">10.2196/12109</pub-id><pub-id pub-id-type="medline">31066686</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bacchi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gluck</surname><given-names>S</given-names> </name><name name-style="western"><surname>Koblar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jannes</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kleinig</surname><given-names>T</given-names> </name></person-group><article-title>Automated information extraction from free-text medical documents for stroke key performance indicators: a pilot study</article-title><source>Intern Med J</source><year>2022</year><month>02</month><volume>52</volume><issue>2</issue><fpage>315</fpage><lpage>317</lpage><pub-id pub-id-type="doi">10.1111/imj.15678</pub-id><pub-id pub-id-type="medline">35187820</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Gaschi</surname><given-names>F</given-names> </name><name name-style="western"><surname>Fontaine</surname><given-names>X</given-names> </name><name name-style="western"><surname>Rastin</surname><given-names>P</given-names> </name><name name-style="western"><surname>Toussaint</surname><given-names>Y</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ben Abacha</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>K</given-names> </name><name name-style="western"><surname>Rumshisky</surname><given-names>A</given-names> </name></person-group><article-title>Multilingual clinical NER: translation or cross-lingual transfer?</article-title><conf-name>Proceedings of the 5th Clinical Natural Language Processing Workshop</conf-name><conf-date>Jul 14, 2023</conf-date><conf-loc>Toronto, Canada</conf-loc><fpage>289</fpage><lpage>311</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.clinicalnlp-1.34</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>P</given-names> </name><name name-style="western"><surname>Choi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Han</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Leveraging natural language processing for efficient information extraction from breast cancer pathology reports: single-institution study</article-title><source>PLoS ONE</source><year>2025</year><volume>20</volume><issue>2</issue><fpage>e0318726</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0318726</pub-id><pub-id pub-id-type="medline">39965024</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moon</surname><given-names>S</given-names> </name><name name-style="western"><surname>Carlson</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Moser</surname><given-names>ED</given-names> </name><etal/></person-group><article-title>Identifying information gaps in electronic health records by using natural language processing: gynecologic surgery history identification</article-title><source>J Med Internet Res</source><year>2022</year><month>01</month><day>28</day><volume>24</volume><issue>1</issue><fpage>e29015</fpage><pub-id pub-id-type="doi">10.2196/29015</pub-id><pub-id pub-id-type="medline">35089141</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Owens</surname><given-names>D</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>DQ</given-names> </name><name name-style="western"><surname>Dohopolski</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rousseau</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Peterson</surname><given-names>ED</given-names> </name><name name-style="western"><surname>Navar</surname><given-names>AM</given-names> </name></person-group><article-title>Accuracy of large language models to identify stroke subtypes within unstructured electronic health record data</article-title><source>Stroke</source><year>2025</year><month>10</month><volume>56</volume><issue>10</issue><fpage>2966</fpage><lpage>2975</lpage><pub-id pub-id-type="doi">10.1161/STROKEAHA.125.051993</pub-id><pub-id pub-id-type="medline">40709446</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mitchell</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Szepietowski</surname><given-names>P</given-names> </name><name name-style="western"><surname>Howard</surname><given-names>R</given-names> </name><etal/></person-group><article-title>A question-and-answer system to extract data from free-text oncological pathology reports (CancerBERT Network): development study</article-title><source>J Med Internet Res</source><year>2022</year><month>03</month><day>23</day><volume>24</volume><issue>3</issue><fpage>e27210</fpage><pub-id pub-id-type="doi">10.2196/27210</pub-id><pub-id pub-id-type="medline">35319481</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>W</given-names> </name><etal/></person-group><article-title>A survey on hallucination in large language models: principles, taxonomy, challenges, and open questions</article-title><source>ACM Trans Inf Syst</source><year>2025</year><month>03</month><day>31</day><volume>43</volume><issue>2</issue><fpage>1</fpage><lpage>55</lpage><pub-id pub-id-type="doi">10.1145/3703155</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abdelghafour</surname><given-names>MAM</given-names> </name><name name-style="western"><surname>Mabrouk</surname><given-names>M</given-names> </name><name name-style="western"><surname>Taha</surname><given-names>Z</given-names> </name></person-group><article-title>Hallucination mitigation techniques in large language models</article-title><source>IJICIS</source><year>2024</year><month>12</month><day>1</day><volume>24</volume><issue>4</issue><fpage>73</fpage><lpage>81</lpage><pub-id pub-id-type="doi">10.21608/ijicis.2024.336135.1365</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sohn</surname><given-names>J</given-names> </name><name name-style="western"><surname>Park</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>C</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Chiruzzo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ritter</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name></person-group><article-title>Rationale-guided retrieval augmented generation for medical question answering</article-title><conf-name>Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics</conf-name><conf-date>Apr 29 to May 4, 2025</conf-date><conf-loc>Albuquerque, New Mexico</conf-loc><fpage>12739</fpage><lpage>12753</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.naacl-long.635</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Khatibi</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rahmani</surname><given-names>AM</given-names> </name></person-group><article-title>MedCoT-RAG: causal chain-of-thought RAG for medical question answering</article-title><access-date>2026-07-16</access-date><conf-name>2025 IEEE 21st International Conference on Body Sensor Networks (BSN)</conf-name><conf-date>Nov 3-5, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/document/11337389">https://ieeexplore.ieee.org/document/11337389</ext-link></comment></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>E</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tenison</surname><given-names>I</given-names> </name><name name-style="western"><surname>Kagal</surname><given-names>L</given-names> </name></person-group><article-title>MediRAG: secure question answering for healthcare data</article-title><conf-name>2024 IEEE International Conference on Big Data (BigData)</conf-name><conf-date>Dec 15-18, 2024</conf-date><conf-loc>Washington, DC, USA</conf-loc><fpage>6476</fpage><lpage>6485</lpage><pub-id pub-id-type="doi">10.1109/BigData62323.2024.10825604</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ning</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>H</given-names> </name></person-group><article-title>MedTrust-RAG: evidence verification and trust alignment for biomedical question answering</article-title><conf-name>2025 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name><conf-date>Dec 15-18, 2025</conf-date><pub-id pub-id-type="doi">10.1109/BIBM66473.2025.11356290</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abo El-Enen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Saad</surname><given-names>S</given-names> </name><name name-style="western"><surname>Nazmy</surname><given-names>T</given-names> </name></person-group><article-title>A survey on retrieval-augmentation generation (RAG) models for healthcare applications</article-title><source>Neural Comput &#x0026; Applic</source><year>2025</year><month>11</month><volume>37</volume><issue>33</issue><fpage>28191</fpage><lpage>28267</lpage><pub-id pub-id-type="doi">10.1007/s00521-025-11666-9</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Rajpurkar</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lopyrev</surname><given-names>K</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>P</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Su</surname><given-names>J</given-names> </name><name name-style="western"><surname>Duh</surname><given-names>K</given-names> </name><name name-style="western"><surname>Carreras</surname><given-names>X</given-names> </name></person-group><article-title>SQuAD: 100,000+ questions for machine comprehension of text</article-title><conf-name>Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 1-5, 2016</conf-date><conf-loc>Austin, Texas</conf-loc><fpage>2383</fpage><lpage>2392</lpage><pub-id pub-id-type="doi">10.18653/v1/D16-1264</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pampari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Raghavan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>J</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Riloff</surname><given-names>E</given-names> </name><name name-style="western"><surname>Chiang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hockenmaier</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tsujii</surname><given-names>J</given-names> </name></person-group><article-title>EmrQA: a large corpus for question answering on electronic medical records</article-title><conf-name>Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Oct 31 to Nov 4, 2018</conf-date><conf-loc>Brussels, Belgium</conf-loc><fpage>2357</fpage><lpage>2368</lpage><pub-id pub-id-type="doi">10.18653/v1/D18-1258</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lanz</surname><given-names>V</given-names> </name><name name-style="western"><surname>Pecina</surname><given-names>P</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Demner-Fushman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Miwa</surname><given-names>M</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tsujii</surname><given-names>J</given-names> </name></person-group><article-title>Paragraph retrieval for enhanced question answering in clinical documents</article-title><conf-name>Proceedings of the 23rd Workshop on Biomedical Natural Language Processing</conf-name><conf-date>Aug 16, 2024</conf-date><conf-loc>Bangkok, Thailand</conf-loc><fpage>580</fpage><lpage>590</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.bionlp-1.48</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yue</surname><given-names>X</given-names> </name><name name-style="western"><surname>Jimenez Gutierrez</surname><given-names>B</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>H</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Jurafsky</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schluter</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tetreault</surname><given-names>J</given-names> </name></person-group><article-title>Clinical reading comprehension: a thorough analysis of the emrQA dataset</article-title><year>2020</year><conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</conf-name><conf-loc>Online</conf-loc><fpage>4474</fpage><lpage>4486</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.410</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="web"><source>Registry of Stroke Care Quality (RES-Q)</source><access-date>2026-02-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.qualityregistry.org">https://www.qualityregistry.org</ext-link></comment></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Rajpurkar</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jia</surname><given-names>R</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>P</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Gurevych</surname><given-names>I</given-names> </name><name name-style="western"><surname>Miyao</surname><given-names>Y</given-names> </name></person-group><article-title>Know what you don&#x2019;t know: unanswerable questions for SQuAD</article-title><conf-name>Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 2</conf-name><conf-date>Jul 15-20, 2018</conf-date><conf-loc>Melbourne, Australia</conf-loc><fpage>784</fpage><lpage>789</lpage><pub-id pub-id-type="doi">10.18653/v1/P18-2124</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Duan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>R</given-names> </name></person-group><article-title>SG-Net: syntax-guided machine reading comprehension</article-title><source>AAAI</source><year>2020</year><volume>34</volume><issue>5</issue><fpage>9636</fpage><lpage>9643</lpage><pub-id pub-id-type="doi">10.1609/aaai.v34i05.6511</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Goodman</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gimpel</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>P</given-names> </name><name name-style="western"><surname>Soricut</surname><given-names>R</given-names> </name></person-group><article-title>ALBERT: a lite BERT for self-supervised learning of language representations</article-title><year>2020</year><access-date>2026-07-15</access-date><conf-name>Proc Int Conf Learn Representations (ICLR 2020)</conf-name><conf-date>Apr 26 to May 1, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=H1eA7AEtvS">https://openreview.net/forum?id=H1eA7AEtvS</ext-link></comment></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>H</given-names> </name></person-group><article-title>Retrospective reader for machine reading comprehension</article-title><source>AAAI</source><year>2021</year><volume>35</volume><issue>16</issue><fpage>14506</fpage><lpage>14514</lpage><pub-id pub-id-type="doi">10.1609/aaai.v35i16.17705</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>P</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>W</given-names> </name></person-group><article-title>DeBERTa: decoding-enhanced BERT with disentangled attention</article-title><access-date>2026-07-15</access-date><conf-name>Proc Int Conf Learn Representations (ICLR)</conf-name><conf-date>May 3-7, 2021</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=XPZIaotutsD">https://openreview.net/forum?id=XPZIaotutsD</ext-link></comment></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Glass</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gliozzo</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chakravarti</surname><given-names>R</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Jurafsky</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schluter</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tetreault</surname><given-names>J</given-names> </name></person-group><article-title>Span selection pre-training for question answering</article-title><conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>Jul 5-10, 2020</conf-date><conf-loc>Online</conf-loc><fpage>2773</fpage><lpage>2782</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.247</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bogireddy</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Dasari</surname><given-names>N</given-names> </name></person-group><article-title>Comparative analysis of ChatGPT-4 and LLaMA: performance evaluation on text summarization, data analysis, and question answering</article-title><conf-name>2024 15th International Conference on Computing Communication and Networking Technologies (ICCCNT)</conf-name><conf-date>Jun 24-28, 2024</conf-date><conf-loc>Kamand, India</conf-loc><fpage>1</fpage><lpage>7</lpage><pub-id pub-id-type="doi">10.1109/ICCCNT61001.2024.10725662</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Beltagy</surname><given-names>I</given-names> </name><name name-style="western"><surname>Peters</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Cohan</surname><given-names>A</given-names> </name></person-group><article-title>Longformer: the long-document transformer</article-title><source>arXiv</source><comment>Preprint posted online on  Apr, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.2004.05150</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zaheer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Guruganesh</surname><given-names>G</given-names> </name><name name-style="western"><surname>Dubey</surname><given-names>KA</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Larochelle</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ranzato</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hadsell</surname><given-names>R</given-names> </name><name name-style="western"><surname>Balcan</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>H</given-names> </name></person-group><article-title>Big bird: transformers for longer sequences</article-title><access-date>2026-07-15</access-date><conf-name>NIPS&#x2019;20: Proceedings of the 34th International Conference on Neural Information Processing Systems</conf-name><conf-date>Dec 6-12, 2020</conf-date><conf-loc>Online</conf-loc><fpage>17283</fpage><lpage>17297</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2020/file/c8512d142a2d849725f31a9a7a361ab9-Paper.pdf">https://proceedings.neurips.cc/paper_files/paper/2020/file/c8512d142a2d849725f31a9a7a361ab9-Paper.pdf</ext-link></comment></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Conneau</surname><given-names>A</given-names> </name><name name-style="western"><surname>Khandelwal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Goyal</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Unsupervised cross-lingual representation learning at scale</article-title><year>2019</year><month>11</month><conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>Jul 5-10, 2020</conf-date><pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.747</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Marone</surname><given-names>M</given-names> </name><name name-style="western"><surname>Weller</surname><given-names>O</given-names> </name><name name-style="western"><surname>Fleshman</surname><given-names>W</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>E</given-names> </name><name name-style="western"><surname>Lawrie</surname><given-names>D</given-names> </name><name name-style="western"><surname>Durme</surname><given-names>B</given-names> </name></person-group><article-title>mmBERT: a modern multilingual encoder with annealed language learning</article-title><source>Arxiv</source><comment>Preprint posted online on  Sep, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2509.06888</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>AQ</given-names> </name><name name-style="western"><surname>Sablayrolles</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mensch</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Mistral 7B</article-title><source>Arxiv</source><comment>Preprint posted online on  Nov, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2310.06825</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Abdin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Aneja</surname><given-names>J</given-names> </name><name name-style="western"><surname>Awadalla</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Phi-3 technical report: a highly capable language model locally on your phone</article-title><source>Arxiv</source><comment>Preprint posted online on  Apr, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2404.14219</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kamath</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ferret</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pathak</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Gemma 3 technical report</article-title><source>Arxiv</source><comment>Preprint posted online on  Mar, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2503.19786</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Labrak</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Bazoge</surname><given-names>A</given-names> </name><name name-style="western"><surname>Morin</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gourraud</surname><given-names>PA</given-names> </name><name name-style="western"><surname>Rouvier</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dufour</surname><given-names>R</given-names> </name></person-group><article-title>BioMistral: a collection of open-source pretrained large language models for medical domains</article-title><year>2024</year><conf-name>Findings of the Association for Computational Linguistics ACL 2024</conf-name><conf-date>Aug 11-16, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.findings-acl.348</pub-id></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Corbeil</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Dada</surname><given-names>A</given-names> </name><name name-style="western"><surname>Attendu</surname><given-names>JM</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Che</surname><given-names>W</given-names> </name><name name-style="western"><surname>Nabende</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shutova</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pilehvar</surname><given-names>MT</given-names> </name></person-group><article-title>A modular approach for clinical slms driven by synthetic data with pre-instruction tuning, model merging, and clinical-tasks alignment</article-title><conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1)</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><conf-loc>Vienna, Austria</conf-loc><fpage>19352</fpage><lpage>19374</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.950</pub-id></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Sellergren</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kazemzadeh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jaroensri</surname><given-names>T</given-names> </name><etal/></person-group><article-title>MedGemma technical report</article-title><source>Arxiv</source><comment>Preprint posted online on  Jul, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2507.05201</pub-id></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wallis</surname><given-names>P</given-names> </name><etal/></person-group><article-title>LoRA: low-rank adaptation of large language models</article-title><access-date>2026-07-15</access-date><conf-name>Proc Int Conf Learn Representations (ICLR)</conf-name><conf-date>Apr 25-29, 2022</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=nZeVKeeFYf9">https://openreview.net/forum?id=nZeVKeeFYf9</ext-link></comment></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>B</given-names> </name><name name-style="western"><surname>Bing</surname><given-names>L</given-names> </name><name name-style="western"><surname>Joty</surname><given-names>S</given-names> </name><name name-style="western"><surname>Si</surname><given-names>L</given-names> </name><name name-style="western"><surname>Miao</surname><given-names>C</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Zong</surname><given-names>C</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>F</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><name name-style="western"><surname>Navigli</surname><given-names>R</given-names> </name></person-group><article-title>MulDA: a multilingual data augmentation framework for low-resource cross-lingual NER</article-title><conf-name>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)</conf-name><conf-date>Aug 1-6, 2021</conf-date><conf-loc>Online</conf-loc><fpage>5834</fpage><lpage>5846</lpage><pub-id pub-id-type="doi">10.18653/v1/2021.acl-long.453</pub-id></nlm-citation></ref><ref id="ref81"><label>81</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bornea</surname><given-names>M</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rosenthal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Florian</surname><given-names>R</given-names> </name><name name-style="western"><surname>Sil</surname><given-names>A</given-names> </name></person-group><article-title>Multilingual transfer learning for QA using translation as data augmentation</article-title><source>AAAI</source><year>2021</year><volume>35</volume><issue>14</issue><fpage>12583</fpage><lpage>12591</lpage><pub-id pub-id-type="doi">10.1609/aaai.v35i14.17491</pub-id></nlm-citation></ref><ref id="ref82"><label>82</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Agarwal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ahmad, Jason Ai</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ai</surname><given-names>J</given-names> </name><etal/></person-group><article-title>gpt-oss-120b &#x0026; gpt-oss-20b model card</article-title><source>arXiv</source><year>2025</year><month>08</month><pub-id pub-id-type="doi">10.48550/arXiv.2508.10925</pub-id></nlm-citation></ref><ref id="ref83"><label>83</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Henry</surname><given-names>S</given-names> </name><name name-style="western"><surname>Buchan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Filannino</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stubbs</surname><given-names>A</given-names> </name><name name-style="western"><surname>Uzuner</surname><given-names>O</given-names> </name></person-group><article-title>2018 n2c2 shared task on adverse drug events and medication extraction in electronic health records</article-title><source>J Am Med Inform Assoc</source><year>2020</year><month>01</month><day>1</day><volume>27</volume><issue>1</issue><fpage>3</fpage><lpage>12</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocz166</pub-id><pub-id pub-id-type="medline">31584655</pub-id></nlm-citation></ref><ref id="ref84"><label>84</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alfattni</surname><given-names>G</given-names> </name><name name-style="western"><surname>Belousov</surname><given-names>M</given-names> </name><name name-style="western"><surname>Peek</surname><given-names>N</given-names> </name><name name-style="western"><surname>Nenadic</surname><given-names>G</given-names> </name></person-group><article-title>Extracting drug names and associated attributes from discharge summaries: text mining study</article-title><source>JMIR Med Inform</source><year>2021</year><month>05</month><day>5</day><volume>9</volume><issue>5</issue><fpage>e24678</fpage><pub-id pub-id-type="doi">10.2196/24678</pub-id><pub-id pub-id-type="medline">33949962</pub-id></nlm-citation></ref><ref id="ref85"><label>85</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Bian</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hogan</surname><given-names>WR</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name></person-group><article-title>Clinical concept extraction using transformers</article-title><source>J Am Med Inform Assoc</source><year>2020</year><month>12</month><day>9</day><volume>27</volume><issue>12</issue><fpage>1935</fpage><lpage>1942</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaa189</pub-id><pub-id pub-id-type="medline">33120431</pub-id></nlm-citation></ref><ref id="ref86"><label>86</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bulgarelli</surname><given-names>L</given-names> </name><name name-style="western"><surname>Pollard</surname><given-names>T</given-names> </name><name name-style="western"><surname>Horng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Celi</surname><given-names>LA</given-names> </name></person-group><article-title>MIMIC-IV (version 2.2)</article-title><year>2023</year><pub-id pub-id-type="doi">10.13026/6mm1-ek67</pub-id></nlm-citation></ref><ref id="ref87"><label>87</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kweon</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kwak</surname><given-names>H</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Globerson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mackey</surname><given-names>L</given-names> </name><name name-style="western"><surname>Belgrave</surname><given-names>D</given-names> </name></person-group><article-title>EHRNoteQA: an LLM benchmark for real-world clinical practice using discharge summaries</article-title><conf-name>Advances in Neural Information Processing Systems 37</conf-name><conf-date>Dec 9-14, 2024</conf-date><conf-loc>Vancouver, BC, Canada</conf-loc><fpage>124575</fpage><lpage>124611</lpage><pub-id pub-id-type="doi">10.52202/079017-3958</pub-id></nlm-citation></ref><ref id="ref88"><label>88</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dada</surname><given-names>A</given-names> </name><name name-style="western"><surname>Koras</surname><given-names>O</given-names> </name><name name-style="western"><surname>Bauer</surname><given-names>M</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Demner-Fushman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>D</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>P</given-names> </name></person-group><article-title>MeDiSumQA: patient-oriented question-answer generation from discharge letters</article-title><conf-name>Proceedings of the Second Workshop on Patient-Oriented Language Processing (CL4Health)</conf-name><conf-date>May 4, 2025</conf-date><conf-loc>Albuquerque, New Mexico</conf-loc><fpage>124</fpage><lpage>136</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.cl4health-1.10</pub-id></nlm-citation></ref><ref id="ref89"><label>89</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Kotschenreuther</surname><given-names>K</given-names> </name></person-group><article-title>EHR-DS-QA: a synthetic QA dataset derived from medical discharge summaries for enhanced medical information retrieval systems</article-title><year>2024</year><publisher-name>PhysioNet</publisher-name><pub-id pub-id-type="doi">10.13026/25fx-f706</pub-id></nlm-citation></ref><ref id="ref90"><label>90</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>AEW</given-names> </name><name name-style="western"><surname>Pollard</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>L</given-names> </name><etal/></person-group><article-title>MIMIC-III, a freely accessible critical care database</article-title><source>Sci Data</source><year>2016</year><month>05</month><day>24</day><volume>3</volume><fpage>160035</fpage><pub-id pub-id-type="doi">10.1038/sdata.2016.35</pub-id><pub-id pub-id-type="medline">27219127</pub-id></nlm-citation></ref><ref id="ref91"><label>91</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Soni</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gayen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Demner-Fushman</surname><given-names>D</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Demner-Fushman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Miwa</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tsujii</surname><given-names>J</given-names> </name></person-group><article-title>Overview of the ArchEHR-QA 2025 shared task on grounded question answering from electronic health records</article-title><conf-name>Proceedings of the 24th Workshop on Biomedical Language Processing</conf-name><conf-date>Aug 1, 2025</conf-date><conf-loc>Viena, Austria</conf-loc><fpage>396</fpage><lpage>405</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.bionlp-1.34</pub-id></nlm-citation></ref><ref id="ref92"><label>92</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lanz</surname><given-names>V</given-names> </name><name name-style="western"><surname>Pecina</surname><given-names>P</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Soni</surname><given-names>S</given-names> </name><name name-style="western"><surname>Demner-Fushman</surname><given-names>D</given-names> </name></person-group><article-title>CUNI-a at ArchEHR-QA 2025: do we need giant LLMs for clinical QA?</article-title><conf-name>Proceedings of the 24th Workshop on Biomedical Language Processing (Shared Tasks)</conf-name><conf-date>Aug 1, 2024</conf-date><conf-loc>Vienna, Austria</conf-loc><fpage>27</fpage><lpage>40</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.bionlp-share.4</pub-id></nlm-citation></ref><ref id="ref93"><label>93</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Balmus</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bogdan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Uban</surname><given-names>AS</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Soni</surname><given-names>S</given-names> </name><name name-style="western"><surname>Demner-Fushman</surname><given-names>D</given-names> </name></person-group><article-title>UniBuc-SB at ArchEHR-QA 2025: a resource-constrained pipeline for relevance classification and grounded answer synthesis</article-title><access-date>2026-07-15</access-date><conf-name>Proceedings of the 24th Workshop on Biomedical Language Processing (Shared Tasks)</conf-name><conf-loc>Vienna, Austria</conf-loc><fpage>62</fpage><lpage>68</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.bionlp-share.7</pub-id></nlm-citation></ref><ref id="ref94"><label>94</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Buonocore</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Parimbelli</surname><given-names>E</given-names> </name><name name-style="western"><surname>Tibollo</surname><given-names>V</given-names> </name><name name-style="western"><surname>Napolitano</surname><given-names>C</given-names> </name><name name-style="western"><surname>Priori</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bellazzi</surname><given-names>R</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Juarez</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Marcos</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stiglic</surname><given-names>G</given-names> </name><name name-style="western"><surname>Tucker</surname><given-names>A</given-names> </name></person-group><article-title>A rule-free approach for cardiological registry filling from italian clinical notes with question answering transformers</article-title><source>Artif Intell Med</source><year>2023</year><publisher-name>Springer</publisher-name><fpage>153</fpage><lpage>162</lpage></nlm-citation></ref><ref id="ref95"><label>95</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Schneider</surname><given-names>ETR</given-names> </name><name name-style="western"><surname>de Souza</surname><given-names>JVA</given-names> </name><name name-style="western"><surname>Knafou</surname><given-names>J</given-names> </name><etal/></person-group><article-title>BioBERTpt - a Portuguese neural language model for clinical named entity recognition</article-title><conf-name>Proceedings of the 3rd Clinical Natural Language Processing Workshop</conf-name><conf-date>Nov 19, 2020</conf-date><conf-loc>Online</conf-loc><fpage>65</fpage><lpage>72</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.clinicalnlp-1.7</pub-id></nlm-citation></ref><ref id="ref96"><label>96</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Seinen</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Kors</surname><given-names>JA</given-names> </name><name name-style="western"><surname>van Mulligen</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Rijnbeek</surname><given-names>PR</given-names> </name></person-group><article-title>Annotation-preserving machine translation of English corpora to validate Dutch clinical concept extraction tools</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>08</month><day>1</day><volume>31</volume><issue>8</issue><fpage>1725</fpage><lpage>1734</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae159</pub-id><pub-id pub-id-type="medline">38934643</pub-id></nlm-citation></ref><ref id="ref97"><label>97</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zegers</surname><given-names>M</given-names> </name><name name-style="western"><surname>Veenstra</surname><given-names>GL</given-names> </name><name name-style="western"><surname>Gerritsen</surname><given-names>G</given-names> </name><name name-style="western"><surname>Verhage</surname><given-names>R</given-names> </name><name name-style="western"><surname>van der Hoeven</surname><given-names>HJG</given-names> </name><name name-style="western"><surname>Welker</surname><given-names>GA</given-names> </name></person-group><article-title>Perceived burden due to registrations for quality monitoring and improvement in hospitals: a mixed methods study</article-title><source>Int J Health Policy Manag</source><year>2020</year><month>07</month><volume>PMID</volume><issue>2</issue><fpage>183</fpage><lpage>196</lpage><pub-id pub-id-type="doi">10.34172/ijhpm.2020.96</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Detailed dataset statistics for the resqEQA dataset, including report and evidence lengths and answer type distributions across six languages.</p><media xlink:href="jmir_v28i1e96347_app1.pdf" xlink:title="PDF File, 56 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Full S1 and S2 results across all training configurations, models, and languages, including possible and impossible instances.</p><media xlink:href="jmir_v28i1e96347_app2.pdf" xlink:title="PDF File, 202 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Performance by answer type for S1, S2, and end-to-end QA, reported across languages and split into possible and impossible instances in the T setting.</p><media xlink:href="jmir_v28i1e96347_app3.pdf" xlink:title="PDF File, 68 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Evidence extraction error analysis showing error-type distribution, dependence on number of gold spans, qualitative span-level errors, and weak correlation with report length across languages.</p><media xlink:href="jmir_v28i1e96347_app4.pdf" xlink:title="PDF File, 178 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Few-shot evaluation results for Answer Prediction (S2) across models and languages, showing accuracy under 0-, 1-, 5-, and 20-shot settings for all, possible, and impossible instances.</p><media xlink:href="jmir_v28i1e96347_app5.pdf" xlink:title="PDF File, 53 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6</label><p>Impact of extended context length on Answer Prediction (S2) performance, showing stable accuracy across increasing report segment lengths (64&#x2013;256 tokens) for all, possible, and impossible instances.</p><media xlink:href="jmir_v28i1e96347_app6.pdf" xlink:title="PDF File, 76 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7</label><p>Source code for reproducing all experiments and analyses.</p><media xlink:href="jmir_v28i1e96347_app7.zip" xlink:title="ZIP File, 623 KB"/></supplementary-material></app-group></back></article>