<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e95198</article-id><article-id pub-id-type="doi">10.2196/95198</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Leveraging Generative Large Language Models for Temporal Relation Extraction From French Clinical Narratives: Prompt-Chaining Approach</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Bannour</surname><given-names>Nesrine</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Assi&#x00E9;</surname><given-names>Guillaume</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jannot</surname><given-names>Anne Sophie</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Boyer</surname><given-names>Olivia</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Garcelon</surname><given-names>Nicolas</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tannier</surname><given-names>Xavier</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Vincent</surname><given-names>Marc</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Clinical Bioinformatics Laboratory, INSERM UMR 1163, Imagine Institute, Universit&#x00E9; Paris Cit&#x00E9;</institution><addr-line>Paris</addr-line><country>France</country></aff><aff id="aff2"><institution>Universit&#x00E9; Paris-Cit&#x00E9;, Imagine Institute, Data Science Platform, INSERM UMR 1163</institution><addr-line>24 Bd du Montparnasse</addr-line><addr-line>Paris</addr-line><country>France</country></aff><aff id="aff3"><institution>Universit&#x00E9; Paris Cit&#x00E9;, CNRS, Inserm, Institut Cochin</institution><addr-line>Paris</addr-line><country>France</country></aff><aff id="aff4"><institution>Department of Endocrinology and National Reference Center for Rare Adrenal Disorders, AP-HP, H&#x00F4;pital Cochin</institution><addr-line>Paris</addr-line><country>France</country></aff><aff id="aff5"><institution>Banque Nationale de Donn&#x00E9;es de Maladies Rares, Assistance Publique &#x2013; H&#x00F4;pitaux de Paris</institution><addr-line>Paris</addr-line><country>France</country></aff><aff id="aff6"><institution>UMRS 1346 &#x2013; HEKA, Universit&#x00E9; Paris Cit&#x00E9;, Inserm, Inria</institution><addr-line>Paris</addr-line><country>France</country></aff><aff id="aff7"><institution>N&#x00E9;phrologie P&#x00E9;diatrique, Centre de R&#x00E9;f&#x00E9;rence MARHEA, H&#x00F4;pital Universitaire Necker-Enfants Malades, Assistance Publique - H&#x00F4;pitaux de Paris (APHP), Imagine Institute, INSERM UMR 1163, Universit&#x00E9; Paris Cit&#x00E9;</institution><addr-line>Paris</addr-line><country>France</country></aff><aff id="aff8"><institution>Sorbonne Universit&#x00E9;, Universit&#x00E9; Sorbonne Paris-Nord, Inserm, Limics</institution><addr-line>Paris</addr-line><country>France</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Castonguay</surname><given-names>Alexandre</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Zheng</surname><given-names>Jiaping</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Chen</surname><given-names>Mei</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Nesrine Bannour, PhD, Universit&#x00E9; Paris-Cit&#x00E9;, Imagine Institute, Data Science Platform, INSERM UMR 1163, 24 Bd du Montparnasse, Paris, 75015, France; <email>nesrine.bannour@institutimagine.org</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>30</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e95198</elocation-id><history><date date-type="received"><day>13</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>06</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>10</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Nesrine Bannour, Guillaume Assi&#x00E9;, Anne Sophie Jannot, Olivia Boyer, Nicolas Garcelon, Xavier Tannier, Marc Vincent. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 30.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e95198"/><abstract><sec><title>Background</title><p>Temporal relation extraction (TRE) in clinical narratives is crucial for understanding patient history, disease progression, and treatment pathways. However, it remains challenging due to limited annotated data and to the complexity of clinical text, including domain-specific terminology, inconsistent information, and implicit temporal reasoning. These challenges are particularly acute for rare diseases, where information such as the date of diagnosis or the onset of key phenotypes is rarely available in structured data, despite being essential for estimating diagnostic delay and reconstructing the natural history of the disease.</p></sec><sec><title>Objective</title><p>We explore the potential of open-weight and on-premises large language models (LLMs) on clinical TRE and normalization using zero- and few-shot prompting, aiming to reduce the time-consuming and costly annotation process. Based on recent promising capabilities of LLMs in understanding and reasoning over text, we evaluate whether these models can efficiently perform TRE and normalization in real-world clinical settings.</p></sec><sec sec-type="methods"><title>Methods</title><p>We cast the temporal relation task as a question-answering task, in which the models extract the temporal expressions associated with a given clinical event. We propose a prompt-chaining strategy that sequentially performs normalization on the extracted temporal expressions. Our LLM-based approach is evaluated on constructed and annotated French real-world clinical narratives, covering temporal relations involving 2 types of clinical events: phenotypes and rare disease diagnoses. Four open-weight LLMs were evaluated across multiple prompt configurations and compared with rule-based and neural baselines using exact-match metrics, an LLM-as-a-judge evaluation, and human validation. We further evaluate our approach by applying it to the English 2012 i2b2 corpus, involving other types of clinical events.</p></sec><sec sec-type="results"><title>Results</title><p>The proposed approach achieves stable extraction performance with only a small number of in-context examples across both clinical event types, reaching a maximum <italic>F</italic><sub>1</sub>-score of 0.72 for rare disease diagnoses and 0.61 for phenotype events. LLM-as-a-judge evaluation supports these findings by capturing minor temporal variations penalized by exact-match metrics, while human validation further confirms the clinical validity of the extracted relations. On the 2012 i2b2 corpus, the approach remains robust despite increased complexity and minimal supervision. Adding normalization using the prompt-chaining strategy maintains stable overall performance, indicating minimal error propagation and efficient temporal normalization.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Prompting LLMs with a minimal set of in-context examples can achieve strong performance for TRE and normalization in low-resource settings with limited domain-specific annotations. Relying on open-weight, on-premises models, the proposed approach therefore offers a practical path toward deployment in real-world clinical workflows and could further be extended to extract key clinical information, such as age at diagnosis or age at onset, particularly in the context of rare diseases, where it could help prefill structured reporting forms, reducing the manual effort needed to retrieve such information directly from clinical narratives.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>electronic health records</kwd><kwd>temporal relation extraction</kwd><kwd>information extraction</kwd><kwd>prompting engineering</kwd><kwd>clinical text</kwd><kwd>rare diseases</kwd><kwd>phenotype</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background and Motivation</title><p>Electronic health records are widely regarded as having significant potential to improve clinical research. However, most data contained in electronic health records are in free-text format. Unstructured text contains up to 80% of crucial clinical information, including temporal information [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Several temporal information extraction approaches have been proposed in recent years to facilitate access to this information, which can help better understand the patient&#x2019;s prior medical history, disease progression, treatment efficacy, etc. Temporal information extraction includes detecting events, identifying temporal expressions, and extracting temporal relations between them.</p><p>In the context of rare diseases, extracting specific temporal information such as the date of diagnosis and the dates of key phenotypes is crucial. Such temporal anchors enable the reconstruction of patient care pathways, support prognostic assessments, and allow researchers to characterize the natural history of rare diseases that are often poorly documented. This information is also required for the curation of rare disease registries and clinical research databases, where structured entries are derived from clinical narratives for reporting and research purposes. In practice, clinicians often need to retrospectively review these narratives to identify and structure relevant temporal information. Automating this process will facilitate the construction of longitudinal patient timelines, improve registry completeness, and reduce the burden of manual reporting.</p></sec><sec id="s1-2"><title>Prior Work in Clinical Temporal Relation Extraction</title><p>There has been a significant interest in temporal relation extraction (TRE) in the clinical domain through the 2012 i2b2 challenge [<xref ref-type="bibr" rid="ref4">4</xref>] and the 2015&#x2010;2017 Clinical TempEval shared tasks [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. Clinical TRE methods evolved from rule-based [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>] to feature-based machine learning [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref13">13</xref>] and neural-based methods [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>Early systems relied on manually defined rules [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref20">20</xref>], which required human expertise and limited generalizability across domains. Following rule-based systems, a range of feature-based machine learning methods has been developed, using, for instance, support vector machine classifiers and conditional random fields, leveraging a range of syntactic, lexical, and semantic features [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Tourille et al [<xref ref-type="bibr" rid="ref12">12</xref>] evaluated random forest and support vector machine models to tackle TRE for clinical corpora in both English and French. Some research studies introduced a hybrid approach that integrates rule-based techniques with a maximum entropy classifier and evaluated its performance on the 2012 i2b2 dataset [<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Neural approaches offered significant improvement to clinical TRE, outperforming traditional feature-based models. Early deep learning methods used convolutional neural networks (CNNs) [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref22">22</xref>] and long short-term memory networks [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>]. Subsequently, the emergence of pretrained transformer-based models [<xref ref-type="bibr" rid="ref25">25</xref>] has driven several research efforts on clinical TRE [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>], applied to benchmark datasets such as THYME [<xref ref-type="bibr" rid="ref28">28</xref>] and 2012 i2b2. Han et al [<xref ref-type="bibr" rid="ref16">16</xref>] introduced OpenNRE, an open and extensible toolkit that implements neural relation extraction models using diverse encoding architectures, such as CNNs, piecewise convolutional neural networks (PCNNs) [<xref ref-type="bibr" rid="ref29">29</xref>], and recurrent neural networks. Other studies have further explored the use of graph neural networks for document-level TRE, aiming to model long-range dependencies between events and temporal expressions. For instance, GRAPHTREX was introduced as a hybrid approach combining span-level entity representations with a heterogeneous graph to model both local and global dependencies [<xref ref-type="bibr" rid="ref30">30</xref>].</p></sec><sec id="s1-3"><title>Challenges in Clinical TRE</title><p>Despite notable progress, TRE in clinical texts remains challenging. To obtain high-performing supervised temporal relation extractors, large amounts of annotated corpora are required. However, the annotation process is known to be time-consuming and highly expensive, particularly in the clinical domain, due to domain expertise need. Consequently, the available datasets have low interannotator agreement [<xref ref-type="bibr" rid="ref31">31</xref>], and annotated corpora are limited for non-English languages. Only a few research works have been conducted on French corpora [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. Moreover, clinical narratives often contain implicit temporal reasoning, domain-specific terminology, and long-distance relations, which continue to limit model performance in real-world applications [<xref ref-type="bibr" rid="ref34">34</xref>].</p></sec><sec id="s1-4"><title>Large Language Models for Clinical TRE</title><p>With the emergence of generative large language models (LLMs), showing promising results in a variety of natural language processing (NLP) tasks, a few studies have evaluated their potential on clinical TRE. Generative LLMs, such as BART [<xref ref-type="bibr" rid="ref35">35</xref>], T5 [<xref ref-type="bibr" rid="ref36">36</xref>], GPT [<xref ref-type="bibr" rid="ref37">37</xref>], and Llama models [<xref ref-type="bibr" rid="ref38">38</xref>], have been used for TRE by casting the task as a text-to-text problem. Different input or output representations impact the performance of encoder-decoder models (T5 and BART) on the Clinical TempEval 2016 dataset, showing that prompting one event at a time yields the best performance [<xref ref-type="bibr" rid="ref39">39</xref>]. Saiz and Altuna [<xref ref-type="bibr" rid="ref40">40</xref>] fine-tuned Relation Extraction By End-to-end Language generation [<xref ref-type="bibr" rid="ref41">41</xref>], a BART-based sequence-to-sequence model originally designed for general TRE, on the 2012 i2b2 corpus, achieving competitive results. Some research work explored the use of ChatGPT in zero-shot and few-shot settings for the TRE task [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref43">43</xref>], reporting that its performance remains limited for long-distance dependencies. A recent review study [<xref ref-type="bibr" rid="ref44">44</xref>] stated that generative LLMs are not widely used for clinical TRE. Furthermore, most efficient models are offered as privately owned services, limiting their use with sensitive data. The advent of open-weight models, most notably the Llama model family [<xref ref-type="bibr" rid="ref38">38</xref>], has facilitated the use of LLMs on real-world data. A new generation of built-in reasoning LLMs has been released, such as DeepSeek-R1 [<xref ref-type="bibr" rid="ref45">45</xref>] and Qwen3 [<xref ref-type="bibr" rid="ref46">46</xref>], to enhance complex task solving.</p><p>Research efforts on open-weight LLMs for clinical TRE showed that zero-shot performance is generally lower than supervised approaches and that temporal consistency remains challenging [<xref ref-type="bibr" rid="ref47">47</xref>]. An LLM-based method combining automated prompt optimization with few-shot in-context learning to predict temporal relations [<xref ref-type="bibr" rid="ref48">48</xref>] was proposed in the 2024 ChemoTimelines shared task [<xref ref-type="bibr" rid="ref49">49</xref>], noting that although fine-tuned smaller LLMs perform best, few-shot prompting can produce reasonable results in low-resource scenarios. Recently, Andrew et al [<xref ref-type="bibr" rid="ref33">33</xref>] evaluated LLMs for extracting temporal relations from French pediatric clinical reports, showing that few-shot prompting with simplified binary labels outperforms multiclass approaches. Their results highlight the potential of LLMs for clinical timeline construction while handling sensitive data securely.</p><p>Latest research has examined prompt-tuning and prompt-learning as efficient alternatives to full fine-tuning in low-resource clinical relation extraction [<xref ref-type="bibr" rid="ref50">50</xref>-<xref ref-type="bibr" rid="ref52">52</xref>], though only a few have focused on clinical TRE [<xref ref-type="bibr" rid="ref53">53</xref>]. An instruction fine-tuning method obtained the highest results [<xref ref-type="bibr" rid="ref54">54</xref>] in the 2024 ChemoTimelines shared task, using the Flan-T5 model [<xref ref-type="bibr" rid="ref55">55</xref>]. For the 2025 edition [<xref ref-type="bibr" rid="ref56">56</xref>], Zhao and Vydiswaran [<xref ref-type="bibr" rid="ref57">57</xref>] combined supervised fine-tuning of Llama models for temporal relation classification with zero-shot Qwen3 prompting for normalization and rule-based postprocessing, obtaining a second-place ranking. As described by Zaghir et al [<xref ref-type="bibr" rid="ref58">58</xref>], there is a significant lack of open-source and locally deployable LLM-based approaches for real-world clinical applications, particularly for French.</p></sec><sec id="s1-5"><title>Contributions</title><p>In this paper, we use open-weight on-premises generative LLMs to extract temporal relations from French clinical narratives in zero-shot and few-shot settings, relying solely on in-context learning [<xref ref-type="bibr" rid="ref59">59</xref>], that is, the ability of the model to learn from directly embedded instructions and examples in the prompt at inference time. Our study focuses on low-resource scenarios, where only few annotated examples are available. Our main goal is to extract the temporal expressions associated with clinical events. To this end, we cast the TRE task as a question-answering problem.</p><p>The main contributions of this paper are summarized as follows:</p><list list-type="bullet"><list-item><p>We propose an LLM-based zero- and few-shot prompting approach that recasts TRE as question answering, identifying the temporal expressions associated with a given clinical event. The extracted expressions are then normalized to a standardized format using a prompt-chaining strategy. To build and evaluate our LLM-based approach, we construct and annotate a real-world corpus of clinical reports written in French, covering temporal relations for phenotypes and rare disease diagnoses, using a defined annotation guideline.</p></list-item><list-item><p>We evaluate 4 open-weight on-premises LLMs, including built-in reasoning models, of comparable parameter sizes in both zero-shot and few-shot settings, comparing them to a rule-based baseline method and neural methods based on the OpenNRE framework.</p></list-item><list-item><p>We assess the performance of our approach using exact-match extraction metrics and go beyond them by incorporating LLM-as-a-judge evaluation to capture subtle temporal variations, complemented with human validation to ensure clinical correctness.</p></list-item><list-item><p>We further evaluate our method on the publicly available English 2012 i2b2 dataset to enable comparison with subsequently developed methods and to assess its robustness for extracting temporal information from clinical narratives in diverse clinical settings.</p></list-item></list></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Task Formulation</title><p>TRE involves identifying and classifying the relation between pairs of events, pairs of temporal expressions, or between an event and a temporal expression. In this work, we are interested in extracting the temporal relations between events and temporal expressions. As illustrated in the left panel of <xref ref-type="fig" rid="figure1">Figure 1</xref>, traditionally, the task is addressed by first constructing all possible event-temporal expression pairs, selecting relevant candidate pairs, and then classifying the relation between each pair. As shown in the right panel of <xref ref-type="fig" rid="figure1">Figure 1</xref>, we reformulated this task as a question-answering problem. Given a clinical text and a clinical event, the objective is to identify and extract all temporal expressions within the text that are associated with the event. If no such temporal expressions exist, the model should return &#x201C;None&#x201D;. Our aim with this task definition and the focus on a single relation type is to develop a reproducible and practical method suitable for real-world clinical use. This reformulation combines the selection of relevant temporal expressions and relation identification into a unified task. Moreover, we used a prompt-chaining strategy to perform normalization on the extracted temporal expressions.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Temporal relation extraction: traditional formulation (left) versus our question-answering reformulation with normalization step (right). The left panel depicts the traditional pipeline of candidates&#x2019; enumeration, filtering, and classification. The right panel illustrates our reformulation, where a question derived from a clinical event is used to query a large language model to extract all associated temporal expressions. The dashed box indicates the normalization step.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95198_fig01.png"/></fig></sec><sec id="s2-2"><title>Prompt Design</title><sec id="s2-2-1"><title>Overview</title><p>The prompt design follows a sequential prompt-chaining framework with 2 distinct prompts: one for temporal extraction and a second for normalizing the extracted expressions. Temporal normalization refers to the process of converting temporal expressions in text into a standardized format, such as YYYY-MM-DD. The same LLMs were used for both steps, forming a pipeline in which the outputs of the extraction prompt are fed directly into the normalization prompt.</p></sec><sec id="s2-2-2"><title>Step 1: Temporal Extraction</title><p>The LLMs were prompted in French to answer a standardized question for each clinical event: &#x201C;What are the dates of the mention of the entity {clinical_event} in the text?&#x201D;</p><p>The expected output was either a list of temporal expressions or a list containing &#x201C;None&#x201D; when no date is associated with the queried event. As shown in <xref ref-type="fig" rid="figure2">Figure 2</xref> (English version of the prompt), each prompt is organized into 3 components: a task description, specifications for a structured output, and a set of explicit instructions. We evaluated the following 4 prompt configurations:</p><list list-type="bullet"><list-item><p>Text: the model is prompted with the full clinical text and the associated question, without any guidance.</p></list-item><list-item><p>Text+rationale: the model is prompted with the same input as in the Text configuration but is additionally instructed to generate 1 or 2 sentences of rationale before producing the final answer.</p></list-item><list-item><p>Text+givenDates: in addition to the input used in the Text configuration, the model is provided with the list of all temporal expressions mentioned in the document (including &#x201C;None&#x201D;), obtained through a rule-based extraction method with additional human review, and is instructed to limit its answer on this list.</p></list-item><list-item><p>Text+rationale+givenDates: the model receives the same inputs as in the Text+givenDates configuration and is further instructed to generate a brief rationale prior to the generation of the final answer.</p></list-item></list><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>The main structure of the extraction prompt.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95198_fig02.png"/></fig></sec><sec id="s2-2-3"><title>Step 2: Temporal Normalization</title><p>The LLMs were prompted to normalize the list of extracted temporal expressions from the step before. Temporal normalization is challenging due to ambiguous temporal expressions, relative dates such as &#x201C;three weeks ago&#x201D; or &#x201C;in two days&#x201D; that require a reference time, and diverse formats. In our work, the LLMs were prompted to normalize relative dates using the report date. Similar to step 1, the normalization prompt is composed of different components to ensure consistent normalization of the extracted temporal expressions into the format YYYY-MM-DD, as illustrated in <xref ref-type="fig" rid="figure3">Figure 3</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>The main structure of the normalization prompt.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95198_fig03.png"/></fig><p>We applied the same model sequentially to the extraction and normalization steps, resulting in a prompt-chaining pipeline that preserves dependencies between tasks. In the few-shot setting, a fixed number of annotated examples were included as in-context demonstrations for each subtask. The number of demonstrations was determined through iterative prompt refinement on the development set, progressively adding examples based on error analysis and targeting challenging cases such as long-range event-temporal dependencies and ambiguous temporal references, until performance reached a satisfactory level. This resulted in 14 demonstrations for phenotype events, 6 demonstrations for rare disease diagnoses in the extraction prompt, and 6 demonstrations for both event types in the temporal normalization prompt. These demonstrations were shared across all models and input configurations within each task.</p></sec></sec><sec id="s2-3"><title>Dataset</title><sec id="s2-3-1"><title>Corpus Construction</title><p>Our work was done in the context of the C&#x2019;IL-LICO research project on ciliopathies. This project and its study protocol were approved by the French National Ethics and Scientific Committee for Research, Studies and Evaluations in the Field of Health (approval 2201437). The data processing was approved by the French Data Protection Authority (CNIL) with a waiver of informed consent under number DR-2023&#x2010;017//920398v1.</p><p>We constructed a corpus of 85 deidentified French patient clinical reports (hospitalization and consultation notes) extracted from Dr. Warehouse, the French clinical data warehouse at Necker Hospital [<xref ref-type="bibr" rid="ref60">60</xref>]. Clinical reports were chosen to ensure sufficient representation of the 2 clinical events of interest: rare disease diagnoses and phenotypes. We annotated 22 documents to create the prompt examples, 21 documents for validating and refining the prompting strategy, and 42 test documents for final evaluation of our LLM-based approach.</p></sec><sec id="s2-3-2"><title>Clinical Events and Temporal Expression Annotation</title><p>Clinical events were initially preannotated using a dictionary-based approach with exact string matching between the clinical text and a terminology-driven list of terms. Rare disease diagnoses were annotated using a filtered Orphanet-based dictionary derived from the 2023 French Orphanet nomenclature file [<xref ref-type="bibr" rid="ref61">61</xref>], excluding diseases classified as nonrare in Europe, obsolete disease names, and terms corresponding to pathology groups. Phenotypes were annotated using a custom dictionary built from the French translation of the Human Phenotype Ontology [<xref ref-type="bibr" rid="ref62">62</xref>] and its synonyms, mapped to the Unified Medical Language System concepts and filtered to retain only Disorders semantic group terms from the 2023AA release [<xref ref-type="bibr" rid="ref63">63</xref>], which includes semantic types relevant to diseases, abnormalities, and clinical findings. Both the Orphanet- and Human Phenotype Ontology&#x2013;based dictionaries were further filtered to exclude terms shorter than 4 characters to reduce noise. The used dictionary-based approach was built using the Entrep&#x00F4;t de donn&#x00E9;es de sant&#x00E9;-natural language processing (EDS-NLP) framework [<xref ref-type="bibr" rid="ref64">64</xref>], which includes built-in support for contextual qualifiers such as negation, hypothesis, and family history. The resulting preannotations were subsequently manually validated by 2 medical experts, who corrected erroneous matches and added missing terms. Overall, as shown in <xref ref-type="table" rid="table1">Table 1</xref>, a total of 488 phenotype mentions and 81 rare disease mentions were annotated. Temporal expressions were detected and normalized using the eds.dates [<xref ref-type="bibr" rid="ref65">65</xref>] component of EDS-NLP. For relative temporal expressions, normalization was performed using the document date, which is encoded in the clinical report filename as recorded in the data warehouse.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Test data statistics per type of event.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Event type</td><td align="left" valign="bottom">Documents, n</td><td align="left" valign="bottom">Event mentions, n</td><td align="left" valign="bottom">Relations, n</td><td align="left" valign="bottom">&#x201C;Doctime&#x201D; relations, n</td><td align="left" valign="bottom">&#x201C;None&#x201D; relations, n</td></tr></thead><tbody><tr><td align="left" valign="top">Phenotype</td><td align="left" valign="top">42</td><td align="left" valign="top">488</td><td align="left" valign="top">356</td><td align="left" valign="top">163</td><td align="left" valign="top">83</td></tr><tr><td align="left" valign="top">Rare disease</td><td align="left" valign="top">42</td><td align="left" valign="top">81</td><td align="left" valign="top">70</td><td align="left" valign="top">40</td><td align="left" valign="top">4</td></tr><tr><td align="left" valign="top">Clinical concept (i2b2)</td><td align="left" valign="top">120</td><td align="left" valign="top">9767</td><td align="left" valign="top">10,087</td><td align="left" valign="top">1860</td><td align="left" valign="top">4079</td></tr></tbody></table></table-wrap></sec><sec id="s2-3-3"><title>Temporal Relation Annotation</title><p>In addition to clinical events and temporal expressions, we manually annotated the document date for each report, defined as the consultation date, admission date, or the report&#x2019;s writing date when the former are unavailable. In specific cases, such as day hospital admission reports introduced by expressions like &#x201C;Day hospital admission report&#x201D; followed by &#x201C;Treatment session on,&#x201D; the document date is defined as the date indicated by the treatment session, as the described events are reported relative to it. Temporal relations are annotated by assigning an &#x201C;is_the_date_of&#x201D; relation between a clinical event and an associated temporal expression when such a relation is present. Preannotated temporal expressions and their normalized values are manually corrected only when involved in an annotated temporal relation, while all other temporal expressions remain unchanged. <xref ref-type="table" rid="table1">Table 1</xref> presents the number of annotated relations for rare disease diagnoses and phenotype events. It includes relations linking events to the document date (&#x201C;Doctime&#x201D; relations) and &#x201C;None&#x201D; relations, which represent cases where no temporal expression is associated with any mention of the event.</p><p>For events with multiple mentions, only the mentions linked to temporal expressions are retained, which is reflected in the counts reported in <xref ref-type="table" rid="table1">Table 1</xref>. Annotation was conducted using the BRAT annotation tool [<xref ref-type="bibr" rid="ref66">66</xref>]. Negated event mentions and family-related events are not included and are therefore not considered when querying the LLMs.</p></sec><sec id="s2-3-4"><title>Adaptation of the 2012 i2b2 Corpus</title><p>For comparison, we evaluated our TRE approach on the 2012 i2b2 corpus [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref67">67</xref>], which includes 310 discharge summaries written in English, split into a training and validation set of 190 documents and a test set of 120 documents. This corpus was annotated with 4 types of events, 4 types of temporal expressions, and 7 types of temporal relations, which are merged into 3 main categories: BEFORE, OVERLAP, and AFTER. We adapted the corpus to our real-world setting through several preprocessing steps. We first inferred all transitive temporal relations using the challenge&#x2019;s official evaluation scripts and restricted the annotations to event-temporal expression relations. We then retained only clinical concepts (problems, tests, and treatments), excluded duration and frequency temporal expressions, and selected the temporal relation types most consistent with our annotation strategy. Specifically, we kept the BEFORE_OVERLAP, ENDED_BY, and BEGIN_BY relations, and the OVERLAP category, which includes SIMULTANEOUS, OVERLAP, and DURING relations. All relations were renamed &#x201C;is_the_date_of&#x201D; as our initial corpus. <xref ref-type="table" rid="table1">Table 1</xref> describes the number of retained relations in the test set.</p><p>To apply our approach to the i2b2 corpus, all prompts were translated into English, and 10 in-context demonstrations sampled from the training set were used in the few-shot setting. The evaluation focused exclusively on the TRE and does not include the normalization of the extracted temporal expressions. Supplementary details on the character-based distances between clinical events and their associated temporal expressions in the test set are provided in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s2-4"><title>Language Model Configurations</title><p>We evaluated 4 generative LLMs, namely, Llama3.3 with 70.6 billion parameters, Qwen2.5 with 72.7 billion parameters, Qwen3 with 32.8 billion parameters, and DeepSeek-R1 with 32.8 billion parameters, all hosted on an on-premises Ollama server to ensure data privacy. Models were served in an 8-bit quantized format for efficiency without compromising performance, and a default context length of 32,000 tokens is used for all experiments. To minimize the variance in outputs, we used a temperature parameter of 0. For the built-in reasoning models Qwen3 and DeepSeek-R1, the maximum number of generated tokens was limited to 2000 to prevent overly long reasoning.</p></sec><sec id="s2-5"><title>Baseline Methods</title><p>We compare the performance of our LLM-based approach with 2 baseline methods: a proximity-based relation extraction method [<xref ref-type="bibr" rid="ref68">68</xref>] and the deep learning&#x2013;based OpenNRE relation extraction toolkit [<xref ref-type="bibr" rid="ref16">16</xref>]. The following section presents some details regarding these 2 baseline approaches.</p><sec id="s2-5-1"><title>Proximity-Based Relation Extraction Method</title><p>This method is a heuristic approach that infers relations based on the textual distance between entities, assuming that closer entities in the text are more likely to be related. It was originally developed to link medications with their associated attributes, where a source entity (eg, a medication) may be associated with multiple target entities (its attributes), while each target entity is generally linked to a single source entity. This assumption is less strictly applicable in the context of TRE.</p><p>In our work, we adopted the symmetric proximity method, which considers distances in both directions and links each target to the nearest source, making it better suited to the variable placement of temporal expressions in clinical narratives. The maximum distance was empirically set to 600 characters, without sentence-level restrictions due to ambiguous sentence boundaries in clinical reports.</p></sec><sec id="s2-5-2"><title>OpenNRE-Based Method</title><p>This method is an open-source and extensible toolkit for neural relation extraction. It classifies the type of relation between a pair of target entities in a sentence. OpenNRE supports a wide range of neural architectures, including CNN-, recurrent neural network-, and transformer-based models. Although originally designed for sentence-level extraction, we adapt it to clinical narratives by relaxing the sentence constraint and considering broader contexts around event-temporal expression pairs. The goal is to predict whether an &#x201C;is_the_date_of&#x201D; or &#x201C;None&#x201D; relation exists between each event and each temporal expression. We trained several models using different encoders, including CamemBERT, ModernCamemBERT, and CNN-based encoders (CNN and PCNN). For transformer-based encoders, we explored the two available encoder representation strategies: (1) using the special classification [CLS] token embedding to represent the input (CLS-based) and (2) using the embeddings of the entity start positions to represent the entities (entity-based).</p><p>For each encoder and representation strategy, we trained 4 models corresponding to different input configurations:</p><list list-type="bullet"><list-item><p>allPairs+FullText: The full clinical text is provided as input for each pair, and all possible candidate event-temporal expression pairs are considered.</p></list-item><list-item><p>allPairs+Passage: Only the passage between the candidate event and temporal expression is provided as input, while still considering all candidate pairs.</p></list-item><list-item><p>LimitedPairs+FullText: The full text is provided for each pair, but candidate pairs are restricted to those within 2 maximum distances determined from the training set: 930 characters for relations between events and the document date, and 8070 characters for relations between events and other temporal expressions. This strategy reduces the number of &#x201C;None&#x201D; relations in the training set from 2840 to 1188 (approximate reduction of 58.2%) while preserving all &#x201C;is_the_date_of&#x201D; relations, thereby alleviating class imbalance. More details on the character-based distances between clinical events and their associated temporal expressions in the train set are provided in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></list-item><list-item><p>LimitedPairs+Passage: Only the passage between the event and temporal expression is used, and candidate pairs are similarly restricted by the same 2 maximum distances.</p></list-item></list><p>All models were trained jointly on relations involving both rare disease diagnoses and phenotypes and evaluated on the complete set of candidate pairs in the test set to assess their performance in extracting relations between clinical events and temporal expressions.</p></sec></sec><sec id="s2-6"><title>Evaluation Metrics</title><p>We evaluated our methods using the 3 main information extraction metrics: precision, recall, and <italic>F</italic><sub>1</sub>-score, focusing exclusively on the positive class (cases where a date is present, as opposed to &#x201C;None&#x201D;). Predicted dates were compared to gold-standard dates using exact match after lowercasing. Certain answers were allowed according to the prompt instructions: Report_date was counted as correct if it matches the document date, and Birth_date was considered correct for age- or birth-related events. 95% CIs were estimated using the empirical bootstrap method [<xref ref-type="bibr" rid="ref69">69</xref>]. Each test corpus is sampled with replacement 1000 times, and the evaluation metrics were computed for each sample.</p><p>Evaluating LLMs remains challenging due to outputs with multiple valid interpretations, factual inconsistencies, and variability in structured tasks such as information extraction, which standard metrics often fail to capture [<xref ref-type="bibr" rid="ref70">70</xref>,<xref ref-type="bibr" rid="ref71">71</xref>]. Therefore, in addition to this strict evaluation, we performed an LLM-as-a-judge validation to account for minor variations in extracted temporal expressions. We prompted the gpt-oss:20b [<xref ref-type="bibr" rid="ref72">72</xref>] model as an impartial temporal evaluator. For each clinical event, the model receives the clinical text, the list of predicted dates, and the reference dates. Each predicted date is judged as correct if it matches or is equivalent to a reference date at the same level of temporal precision (year, month, or day), and incorrect otherwise. Precision, recall, and <italic>F</italic><sub>1</sub> are then derived from these judgments. This evaluation framework enables scalable and context-aware validation while leveraging the original clinical narrative.</p><p>Finally, we conducted a human-based validation on our best-performing experiments to provide a more relaxed assessment of the LLM outputs. We reviewed each predicted temporal relation and assigned a binary judgment (correct or incorrect) based on the clinical text. This precision metric measures the proportion of model predictions that are true positives among all predicted relations.</p><p>Note that the normalization step was only evaluated under strict exact-match metrics.</p><p>Carbon footprint in terms of CO<sub>2</sub> equivalent of our LLM-based experiments was estimated using GreenAlgorithms (version 3.0; University of Cambridge) [<xref ref-type="bibr" rid="ref73">73</xref>], based on runtime, computing hardware, and location, covering the full extraction and normalization pipeline for our real-world corpus and the extraction stage only for the i2b2 corpus.</p></sec><sec id="s2-7"><title>Ethical Considerations</title><p>This work is part of the C&#x2019;IL-LICO research project on ciliopathies. This project and its study protocol were approved by the French National Ethics and Scientific Committee for Research, Studies and Evaluations in the Field of Health (approval 2201437). The data processing was approved by the French Data Protection Authority (CNIL) with a waiver of informed consent under number DR-2023&#x2010;017//920398v1. The clinical data used in our study were deidentified in accordance with General Data Protection Regulation protocols and regulations, and all our experiments were conducted locally on secure machines. The 2012 i2b2 corpus is available for research purposes under a data use agreement.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>LLM-Based TRE Performance</title><p><xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref> report the exact-match performance with 95% bootstrap CIs across all models, prompting strategies, and input configurations for rare disease diagnoses and phenotype events, respectively. Few-shot prompting consistently outperforms zero-shot extraction for rare disease diagnoses, while for phenotype events, the gains are generally modest or inconsistent across models and input settings.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Exact-match evaluation of zero-shot and few-shot relation extraction performance (rare disease): precision, recall, and <italic>F</italic><sub>1</sub>-score.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom" colspan="3">Llama3.3</td><td align="left" valign="bottom" colspan="3">Qwen2.5</td><td align="left" valign="bottom" colspan="3">Qwen3</td><td align="left" valign="bottom" colspan="3">DeepSeek-R1</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">Precision (95% CI)</td><td align="left" valign="top">Recall (95% CI)</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="top">Precision (95% CI)</td><td align="left" valign="top">Recall (95% CI)</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="top">Precision (95% CI)</td><td align="left" valign="top">Recall (95% CI)</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="top">Precision (95% CI)</td><td align="left" valign="top">Recall (95% CI)</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="13">Zero-shot extraction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text</td><td align="left" valign="top">0.34 (0.23&#x2010;0.51)</td><td align="left" valign="top">0.67 (0.57&#x2010;0.77)</td><td align="left" valign="top">0.45 (0.33&#x2010;0.59)</td><td align="left" valign="top">0.61 (0.49&#x2010;0.73)</td><td align="left" valign="top">0.67 (0.55&#x2010;0.78)</td><td align="left" valign="top">0.64 (0.53&#x2010;0.74)</td><td align="left" valign="top">0.16 (0.12&#x2010;0.22)</td><td align="left" valign="top">0.61 (0.46&#x2010;0.74)</td><td align="left" valign="top">0.25 (0.19&#x2010;0.33)</td><td align="left" valign="top">0.19 (0.08&#x2010;0.44)</td><td align="left" valign="top">0.55 (0.42&#x2010;0.67)</td><td align="left" valign="top">0.29 (0.14&#x2010;0.52)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale</td><td align="left" valign="top">0.35 (0.26&#x2010;0.47)</td><td align="left" valign="top">0.77 (0.67&#x2010;0.87)</td><td align="left" valign="top">0.48 (0.38&#x2010;0.59)</td><td align="left" valign="top">0.55 (0.43&#x2010;0.7)</td><td align="left" valign="top">0.57 (0.45&#x2010;0.69)</td><td align="left" valign="top">0.56 (0.46&#x2010;0.67)</td><td align="left" valign="top">0.48 (0.38&#x2010;0.61)</td><td align="left" valign="top">0.51 (0.4&#x2010;0.63)</td><td align="left" valign="top">0.5 (0.4&#x2010;0.6)</td><td align="left" valign="top">0.55 (0.42&#x2010;0.66)</td><td align="left" valign="top">0.56 (0.41&#x2010;0.67)</td><td align="left" valign="top">0.56 (0.42&#x2010;0.65)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+givenDates</td><td align="left" valign="top">0.38 (0.26&#x2010;0.56)</td><td align="left" valign="top">0.7 (0.59&#x2010;0.8)</td><td align="left" valign="top">0.49 (0.37&#x2010;0.63)</td><td align="left" valign="top">0.61 (0.5&#x2010;0.74)</td><td align="left" valign="top">0.7 (0.59&#x2010;0.79)</td><td align="left" valign="top">0.65 (0.55&#x2010;0.75)</td><td align="left" valign="top">0.3 (0.23&#x2010;0.4)</td><td align="left" valign="top">0.77 (0.67&#x2010;0.87)</td><td align="left" valign="top">0.43 (0.34&#x2010;0.53)</td><td align="left" valign="top">0.13 (0.09&#x2010;0.17)</td><td align="left" valign="top">0.92 (0.86&#x2010;0.98)</td><td align="left" valign="top">0.22 (0.17&#x2010;0.29)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale+givenDates</td><td align="left" valign="top">0.42 (0.32&#x2010;0.57)</td><td align="left" valign="top">0.74 (0.64&#x2010;0.84)</td><td align="left" valign="top">0.54 (0.43&#x2010;0.65)</td><td align="left" valign="top">0.69 (0.57&#x2010;0.81)</td><td align="left" valign="top">0.67 (0.56&#x2010;0.77)</td><td align="left" valign="top">0.68 (0.57&#x2010;0.78)</td><td align="left" valign="top">0.65 (0.54&#x2010;0.79)</td><td align="left" valign="top">0.57 (0.47&#x2010;0.69)</td><td align="left" valign="top">0.61 (0.51&#x2010;0.71)</td><td align="left" valign="top">0.72 (0.59&#x2010;0.84)</td><td align="left" valign="top">0.57 (0.47&#x2010;0.7)</td><td align="left" valign="top">0.64 (0.53&#x2010;0.74)</td></tr><tr><td align="left" valign="top" colspan="13">Few-shot extraction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text</td><td align="left" valign="top">0.41 (0.31&#x2010;0.52)</td><td align="left" valign="top">0.85 (0.74&#x2010;0.92)</td><td align="left" valign="top">0.55 (0.44&#x2010;0.65)</td><td align="left" valign="top">0.58 (0.47&#x2010;0.72)</td><td align="left" valign="top">0.82 (0.73&#x2010;0.9)</td><td align="left" valign="top">0.68 (0.59&#x2010;0.78)</td><td align="left" valign="top">0.48 (0.36&#x2010;0.61)</td><td align="left" valign="top">0.7 (0.59&#x2010;0.81)</td><td align="left" valign="top">0.57 (0.46&#x2010;0.67)</td><td align="left" valign="top">0.46 (0.34&#x2010;0.61)</td><td align="left" valign="top">0.7 (0.6&#x2010;0.8)</td><td align="left" valign="top">0.56 (0.44&#x2010;0.68)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale</td><td align="left" valign="top">0.56 (0.41&#x2010;0.72)</td><td align="left" valign="top">0.73 (0.61&#x2010;0.81)</td><td align="left" valign="top">0.63 (0.51&#x2010;0.73)</td><td align="left" valign="top">0.72 (0.63&#x2010;0.82)</td><td align="left" valign="top">0.73 (0.63&#x2010;0.83)</td><td align="left" valign="top"><italic>0.72 (0.65&#x2010;0.79)</italic><sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="top">0.74 (0.62&#x2010;0.87)</td><td align="left" valign="top">0.61 (0.49&#x2010;0.72)</td><td align="left" valign="top">0.67 (0.56&#x2010;0.77)</td><td align="left" valign="top">0.76 (0.6&#x2010;0.82)</td><td align="left" valign="top">0.64 (0.47&#x2010;0.72)</td><td align="left" valign="top"><italic>0.7 (0.54&#x2010;0.75)</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+givenDates</td><td align="left" valign="top">0.32 (0.21&#x2010;0.49)</td><td align="left" valign="top">0.68 (0.59&#x2010;0.79)</td><td align="left" valign="top">0.44 (0.32&#x2010;0.58)</td><td align="left" valign="top">0.53 (0.42&#x2010;0.67)</td><td align="left" valign="top">0.71 (0.61&#x2010;0.82)</td><td align="left" valign="top">0.61 (0.5&#x2010;0.72)</td><td align="left" valign="top">0.6 (0.49&#x2010;0.71)</td><td align="left" valign="top">0.79 (0.69&#x2010;0.89)</td><td align="left" valign="top"><italic>0.68 (0.58&#x2010;0.77)</italic></td><td align="left" valign="top">0.32 (0.24&#x2010;0.43)</td><td align="left" valign="top">0.68 (0.58&#x2010;0.78)</td><td align="left" valign="top">0.43 (0.35&#x2010;0.54)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale+givenDates</td><td align="left" valign="top">0.76 (0.65&#x2010;0.87)</td><td align="left" valign="top">0.71 (0.61&#x2010;0.81)</td><td align="left" valign="top"><italic>0.73 (0.64&#x2010;0.83)</italic></td><td align="left" valign="top">0.57 (0.44&#x2010;0.73)</td><td align="left" valign="top">0.79 (0.69&#x2010;0.88)</td><td align="left" valign="top">0.66 (0.56&#x2010;0.78)</td><td align="left" valign="top">0.6 (0.42&#x2010;0.85)</td><td align="left" valign="top">0.62 (0.52&#x2010;0.73)</td><td align="left" valign="top">0.61 (0.49&#x2010;0.75)</td><td align="left" valign="top">0.75 (0.62&#x2010;0.88)</td><td align="left" valign="top">0.54 (0.43&#x2010;0.66)</td><td align="left" valign="top">0.63 (0.52&#x2010;0.74)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>For each model, the highest <italic>F</italic><sub>1</sub>-score across input configurations is highlighted in italics format.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Exact-match evaluation of zero-shot and few-shot relation extraction performance (phenotypes): precision, recall, and <italic>F</italic><sub>1</sub>-score.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom" colspan="3">Llama3.3</td><td align="left" valign="bottom" colspan="3">Qwen2.5</td><td align="left" valign="bottom" colspan="3">Qwen3</td><td align="left" valign="bottom" colspan="3">DeepSeek-R1</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">Precision (95% CI)</td><td align="left" valign="top">Recall (95% CI)</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="top">Precision (95% CI)</td><td align="left" valign="top">Recall (95% CI)</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="top">Precision (95% CI)</td><td align="left" valign="top">Recall (95% CI)</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="top">Precision (95% CI)</td><td align="left" valign="top">Recall (95% CI)</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="13">Zero-shot extraction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text</td><td align="left" valign="top">0.44 (0.38&#x2010;0.5)</td><td align="left" valign="top">0.77 (0.71&#x2010;0.81)</td><td align="left" valign="top">0.56 (0.5&#x2010;0.61)</td><td align="left" valign="top">0.48 (0.43&#x2010;0.53)</td><td align="left" valign="top">0.75 (0.69&#x2010;0.8)</td><td align="left" valign="top">0.58 (0.53&#x2010;0.64)</td><td align="left" valign="top">0.2 (0.13&#x2010;0.29)</td><td align="left" valign="top">0.66 (0.6&#x2010;0.72)</td><td align="left" valign="top">0.31 (0.22&#x2010;0.4)</td><td align="left" valign="top">0.24 (0.2&#x2010;0.28)</td><td align="left" valign="top">0.5 (0.44&#x2010;0.56)</td><td align="left" valign="top">0.32 (0.28&#x2010;0.37)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale</td><td align="left" valign="top">0.44 (0.38&#x2010;0.5)</td><td align="left" valign="top">0.75 (0.69&#x2010;0.8)</td><td align="left" valign="top">0.55 (0.5&#x2010;0.61)</td><td align="left" valign="top">0.5 (0.44&#x2010;0.56)</td><td align="left" valign="top">0.65 (0.59&#x2010;0.71)</td><td align="left" valign="top">0.56 (0.51&#x2010;0.62)</td><td align="left" valign="top">0.44 (0.37&#x2010;0.5)</td><td align="left" valign="top">0.49 (0.42&#x2010;0.55)</td><td align="left" valign="top">0.46 (0.4&#x2010;0.52)</td><td align="left" valign="top">0.5 (0.43&#x2010;0.55)</td><td align="left" valign="top">0.57 (0.5&#x2010;0.62)</td><td align="left" valign="top">0.53 (0.47&#x2010;0.58)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+givenDates</td><td align="left" valign="top">0.45 (0.4&#x2010;0.5)</td><td align="left" valign="top">0.74 (0.69&#x2010;0.8)</td><td align="left" valign="top">0.56 (0.51&#x2010;0.61)</td><td align="left" valign="top">0.5 (0.45&#x2010;0.55)</td><td align="left" valign="top">0.75 (0.7&#x2010;0.8)</td><td align="left" valign="top">0.6 (0.55&#x2010;0.65)</td><td align="left" valign="top">0.34 (0.29&#x2010;0.39)</td><td align="left" valign="top">0.77 (0.72&#x2010;0.82)</td><td align="left" valign="top">0.47 (0.42&#x2010;0.52)</td><td align="left" valign="top">0.1 (0.09&#x2010;0.12)</td><td align="left" valign="top">0.85 (0.8&#x2010;0.89)</td><td align="left" valign="top">0.18 (0.16&#x2010;0.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale+givenDates</td><td align="left" valign="top">0.46 (0.42&#x2010;0.51)</td><td align="left" valign="top">0.75 (0.7&#x2010;0.8)</td><td align="left" valign="top">0.57 (0.53&#x2010;0.62)</td><td align="left" valign="top">0.53 (0.47&#x2010;0.59)</td><td align="left" valign="top">0.72 (0.67&#x2010;0.77)</td><td align="left" valign="top"><italic>0.61 (0.56&#x2010;0.66)</italic><sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="top">0.53 (0.47&#x2010;0.59)</td><td align="left" valign="top">0.67 (0.61&#x2010;0.71)</td><td align="left" valign="top"><italic>0.59 (0.53&#x2010;0.64)</italic></td><td align="left" valign="top">0.55 (0.48&#x2010;0.61)</td><td align="left" valign="top">0.63 (0.58&#x2010;0.68)</td><td align="left" valign="top"><italic>0.59 (0.53&#x2010;0.64)</italic></td></tr><tr><td align="left" valign="top" colspan="13">Few-shot extraction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text</td><td align="left" valign="top">0.48 (0.42&#x2010;0.52)</td><td align="left" valign="top">0.72 (0.65&#x2010;0.76)</td><td align="left" valign="top">0.57 (0.51&#x2010;0.61)</td><td align="left" valign="top">0.46 (0.41&#x2010;0.52)</td><td align="left" valign="top">0.64 (0.58&#x2010;0.7)</td><td align="left" valign="top">0.54 (0.48&#x2010;0.59)</td><td align="left" valign="top">0.45 (0.39&#x2010;0.5)</td><td align="left" valign="top">0.57 (0.51&#x2010;0.62)</td><td align="left" valign="top">0.5 (0.44&#x2010;0.55)</td><td align="left" valign="top">0.42 (0.35&#x2010;0.5)</td><td align="left" valign="top">0.23 (0.18&#x2010;0.28)</td><td align="left" valign="top">0.3 (0.24&#x2010;0.35)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale</td><td align="left" valign="top">0.58 (0.51&#x2010;0.64)</td><td align="left" valign="top">0.6 (0.52&#x2010;0.65)</td><td align="left" valign="top"><italic>0.59 (0.53-0.64)</italic></td><td align="left" valign="top">0.54 (0.47&#x2010;0.59)</td><td align="left" valign="top">0.65 (0.58&#x2010;0.7)</td><td align="left" valign="top">0.59 (0.52&#x2010;0.64)</td><td align="left" valign="top">0.59 (0.52&#x2010;0.65)</td><td align="left" valign="top">0.55 (0.49&#x2010;0.61)</td><td align="left" valign="top">0.57 (0.51&#x2010;0.63)</td><td align="left" valign="top">0.57 (0.47&#x2010;0.63)</td><td align="left" valign="top">0.41 (0.34&#x2010;0.45)</td><td align="left" valign="top">0.48 (0.4&#x2010;0.52)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+givenDates</td><td align="left" valign="top">0.44 (0.38&#x2010;0.49)</td><td align="left" valign="top">0.79 (0.74&#x2010;0.83)</td><td align="left" valign="top">0.57 (0.51&#x2010;0.61)</td><td align="left" valign="top">0.4 (0.35&#x2010;0.45)</td><td align="left" valign="top">0.6 (0.55&#x2010;0.67)</td><td align="left" valign="top">0.48 (0.43&#x2010;0.54)</td><td align="left" valign="top">0.43 (0.37&#x2010;0.5)</td><td align="left" valign="top">0.7 (0.64&#x2010;0.75)</td><td align="left" valign="top">0.53 (0.48&#x2010;0.59)</td><td align="left" valign="top">0.24 (0.19&#x2010;0.29)</td><td align="left" valign="top">0.59 (0.52&#x2010;0.64)</td><td align="left" valign="top">0.34 (0.28&#x2010;0.39)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale+givenDates</td><td align="left" valign="top">0.61 (0.54&#x2010;0.67)</td><td align="left" valign="top">0.54 (0.48&#x2010;0.6)</td><td align="left" valign="top">0.57 (0.51&#x2010;0.62)</td><td align="left" valign="top">0.52 (0.47&#x2010;0.59)</td><td align="left" valign="top">0.69 (0.64&#x2010;0.75)</td><td align="left" valign="top">0.6 (0.55&#x2010;0.65)</td><td align="left" valign="top">0.6 (0.53&#x2010;0.66)</td><td align="left" valign="top">0.58 (0.52&#x2010;0.63)</td><td align="left" valign="top"><italic>0.59 (0.53&#x2010;0.63)</italic></td><td align="left" valign="top">0.65 (0.57&#x2010;0.72)</td><td align="left" valign="top">0.43 (0.36&#x2010;0.49)</td><td align="left" valign="top">0.52 (0.45&#x2010;0.57)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>For each model, the highest <italic>F</italic><sub>1</sub>-score across input configurations is highlighted in italics format.</p></fn></table-wrap-foot></table-wrap><p>Configurations including rationale generation provide more balanced precision and recall, while those incorporating candidate temporal expressions achieve higher recall values, particularly when combined with rationale generation. In the zero-shot setting, the Text+rationale+givenDates configuration achieves the best results across all evaluated models for both phenotypes and rare disease diagnoses.</p><p>For rare disease diagnoses, the best results are obtained by Llama3.3 in the few-shot setting Text+rationale+givenDates configuration, achieving an <italic>F</italic><sub>1</sub> of 0.73. Comparable performance is achieved by Qwen2.5, reaching an <italic>F</italic><sub>1</sub> of 0.72 in the few-shot Text+rationale configuration. For phenotype events, overall scores are lower. The best <italic>F</italic><sub>1</sub>-scores are achieved by Qwen2.5 in the zero-shot Text+rationale+givenDates configuration, with an <italic>F</italic><sub>1</sub> of 0.61, and in the few-shot setting with the same configuration, reaching an <italic>F</italic><sub>1</sub> of 0.6. Llama3.3 and Qwen3 also yield competitive performance, with maximum <italic>F</italic><sub>1</sub>-scores of 0.59.</p><p>Across both event types, the reasoning Qwen3 and DeepSeek-R1 models obtain lower <italic>F</italic><sub>1</sub>-scores than Qwen2.5 and Llama3.3, with consistently higher recall and lower precision across most configurations.</p><p><xref ref-type="fig" rid="figure4">Figures 4</xref> and <xref ref-type="fig" rid="figure5">5</xref> illustrate how character-based distance between clinical events and temporal expressions influences extraction performance for rare disease and phenotype events with Qwen2.5 across all input settings. In both figures, the upper panels present <italic>F</italic><sub>1</sub>-scores computed within each distance bin, where distance bins are defined by percentiles of the distance distribution (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The lower panels show the distance distribution using a Gaussian kernel density estimate.</p><p>For rare disease relations, <italic>F</italic><sub>1</sub>-scores are highest at short distances (up to 57 characters: 0.58&#x2010;0.73) and progressively decline across medium distance bins (57&#x2010;159 and 159&#x2010;315 characters: 0.44&#x2010;0.73 <italic>F</italic><sub>1</sub>) with a substantial performance drop at long distances (more than 315 characters: 0.20&#x2010;0.57 <italic>F</italic><sub>1</sub> and <italic>F</italic><sub>1</sub>&#x003C;0.30 for several zero-shot configurations). The Gaussian kernel density estimate curve indicates that most rare disease relations occur within shorter distances, whereas long-distance relations are less frequent but still more challenging.</p><p>For phenotype relations, <italic>F</italic><sub>1</sub>-scores remain relatively stable (0.52&#x2010;0.82) across the first 3 distance bins (13&#x2010;1097 characters) for most configurations. The configuration Text+rationale+givenDates achieved the highest performance (0.82 of <italic>F</italic><sub>1</sub>) in both zero-shot and few-shot settings in the shortest distance range (13&#x2010;97 characters). Performance progressively declined as distance increased, with <italic>F</italic><sub>1</sub>-scores dropping to 0.29&#x2010;0.36 beyond 1999 characters.</p><p>Results obtained using our LLM-as-a-judge validation framework for the TRE are presented in Tables S3 and S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. This assessment allows for minor variations in predicted temporal expressions while maintaining strict temporal consistency. The results closely align with the exact-match evaluation but consistently yield higher <italic>F</italic><sub>1</sub>-scores across all models and event types.</p><p>To assess the reliability of the LLM-as-a-judge framework, we compare its judgments with human evaluations on a subset of examples. Across all configurations using Qwen2.5, the agreement between the LLM and the human annotator reaches an average <italic>F</italic><sub>1</sub> of 0.96 for phenotypes and 1.0 for rare disease events, indicating a very high level of consistency.</p><p><xref ref-type="table" rid="table4">Table 4</xref> reports the proportion of predicted dates judged as correct by a human evaluator for rare disease and phenotype events for our best-performing Qwen2.5 experiments. For rare disease diagnoses, precision ranges from 0.65 to 0.85 across all configurations, and for phenotypes, from 0.6 to 0.71. These values exceed the exact-match precision (<xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>), confirming a high precision of the extracted relations.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Impact of character-based distance between events and temporal expressions on rare disease relation extraction performance using Qwen2.5. Upper panel: <italic>F</italic><sub>1</sub>-scores across character-distance bins. The number (n) indicates the number of relations. Lower panel: Gaussian KDE of the distance distribution. KDE: kernel density estimate.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95198_fig04.png"/></fig><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Impact of character-based distance between events and temporal expressions on phenotype relation extraction performance using Qwen2.5. Upper panel: <italic>F</italic><sub>1</sub>-scores across character-distance bins. The number (n) indicates the number of relations. Lower panel: Gaussian KDE of the distance distribution. KDE: kernel density estimate.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e95198_fig05.png"/></fig><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Human-evaluated precision of model predictions using Qwen2.5<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Rare disease</td><td align="left" valign="bottom">Phenotypes</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Zero-shot extraction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text</td><td align="left" valign="top"><italic>0.85</italic><sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> (0.61)</td><td align="left" valign="top">0.64 (0.48)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale</td><td align="left" valign="top">0.65 (0.55)</td><td align="left" valign="top">0.66 (0.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+givenDates</td><td align="left" valign="top">0.77 (0.61)</td><td align="left" valign="top">0.61 (0.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale+givenDates</td><td align="left" valign="top">0.78 (0.69)</td><td align="left" valign="top">0.66 (0.53)</td></tr><tr><td align="left" valign="top" colspan="3">Few-shot extraction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text</td><td align="left" valign="top">0.74 (0.58)</td><td align="left" valign="top">0.63 (0.46)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale</td><td align="left" valign="top">0.83 (0.72)</td><td align="left" valign="top">0.66 (0.54)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+givenDates</td><td align="left" valign="top">0.72 (0.53)</td><td align="left" valign="top">0.6 (0.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale+givenDates</td><td align="left" valign="top">0.7 (0.57)</td><td align="left" valign="top"><italic>0.71</italic> (0.52)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Values in parentheses indicate the exact-match precision scores presented in <xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>.</p></fn><fn id="table4fn2"><p><sup>b</sup>For each model, the highest <italic>F</italic><sub>1</sub>-score across input configurations is highlighted in italics format.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Baseline Methods Performance</title><p><xref ref-type="table" rid="table5">Table 5</xref> presents the results of the proximity-based relation extraction method on rare disease and phenotype events. The method achieves higher performance on rare disease events, with an <italic>F</italic><sub>1</sub> of 0.48, while performance is lower for phenotypes, with an <italic>F</italic><sub>1</sub> of 0.24. The values in parentheses in <xref ref-type="table" rid="table5">Table 5</xref> correspond to our best-performing LLM, Qwen2.5, which outperforms the proximity-based method across both event types (<xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>).</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Exact-match evaluation of proximity-based relation extraction<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">Rare disease</td><td align="left" valign="top">0.51 (0.53&#x2010;0.72)</td><td align="left" valign="top">0.46 (0.57&#x2010;0.82)</td><td align="left" valign="top">0.48 (0.56&#x2010;0.72)</td></tr><tr><td align="left" valign="top">Phenotypes</td><td align="left" valign="top">0.31 (0.4&#x2010;0.54)</td><td align="left" valign="top">0.19 (0.6&#x2010;0.75)</td><td align="left" valign="top">0.24 (0.48&#x2010;0.61)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>The ranges indicate the minimum-maximum performance of our best-performing LLM, Qwen2.5, across input configurations.</p></fn></table-wrap-foot></table-wrap><p><xref ref-type="table" rid="table6">Table 6</xref> summarizes the performance of the OpenNRE-based models trained on clinical narratives using different encoders (CamemBERT, ModernCamemBERT, CNN, and PCNN), 4 input configurations, and 2 encoder representation strategies for transformer-based encoders, as previously described. Due to the 512-token limit of CamemBERT, which causes truncation of longer clinical texts, we restrict its experiments to passage-based configurations, rather than the full text. Similarly, for the CNN-based encoders, we report results only for the LimitedPairs+Passage configuration.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Exact-match evaluation of OpenNRE-based relation extraction: precision, recall, and <italic>F</italic><sub>1</sub>-score.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom" colspan="3">AllPairs+FullText</td><td align="left" valign="bottom" colspan="3">AllPairs+Passage</td><td align="left" valign="bottom" colspan="3">LimitedPairs+FullText</td><td align="left" valign="bottom" colspan="3">LimitedPairs+Passage</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="13">Rare disease</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ModernCamemBERT[cls]</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.13</td><td align="left" valign="top">0.14</td><td align="left" valign="top">0.72</td><td align="left" valign="top">0.24</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.75</td><td align="left" valign="top">0.13</td><td align="left" valign="top">0.3</td><td align="left" valign="top">0.55</td><td align="left" valign="top">0.39</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ModernCamemBERT[entity]</td><td align="left" valign="top">0.18</td><td align="left" valign="top">0.63</td><td align="left" valign="top"><italic>0.2</italic><sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup></td><td align="left" valign="top">0.19</td><td align="left" valign="top">0.4</td><td align="left" valign="top">0.26</td><td align="left" valign="top">0.18</td><td align="left" valign="top">0.19</td><td align="left" valign="top"><italic>0.18</italic></td><td align="left" valign="top">0.16</td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.26</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CamemBERT[cls]</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.8</td><td align="left" valign="top">0.46</td><td align="left" valign="top"><italic>0.58</italic></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.32</td><td align="left" valign="top">0.26</td><td align="left" valign="top">0.29</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CamemBERT[entity]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.8</td><td align="left" valign="top">0.28</td><td align="left" valign="top">0.41</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.65</td><td align="left" valign="top">0.31</td><td align="left" valign="top"><italic>0.41</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CNN-based[cnnEncoder]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.75</td><td align="left" valign="top">0.04</td><td align="left" valign="top">0.08</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CNN-based[pcnnEncoder]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.21</td><td align="left" valign="top">0.83</td><td align="left" valign="top">0.33</td></tr><tr><td align="left" valign="top" colspan="13">Phenotypes</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ModernCamemBERT[cls]</td><td align="left" valign="top">0.1</td><td align="left" valign="top">0.59</td><td align="left" valign="top">0.1</td><td align="left" valign="top">0.1</td><td align="left" valign="top">0.58</td><td align="left" valign="top">0.18</td><td align="left" valign="top">0.05</td><td align="left" valign="top">0.52</td><td align="left" valign="top">0.09</td><td align="left" valign="top">0.21</td><td align="left" valign="top">0.41</td><td align="left" valign="top">0.27</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ModernCamemBERT[entity]</td><td align="left" valign="top">0.01</td><td align="left" valign="top">0.15</td><td align="left" valign="top"><italic>0.12</italic></td><td align="left" valign="top">0.13</td><td align="left" valign="top">0.34</td><td align="left" valign="top">0.19</td><td align="left" valign="top">0.11</td><td align="left" valign="top">0.17</td><td align="left" valign="top"><italic>0.13</italic></td><td align="left" valign="top">0.13</td><td align="left" valign="top">0.55</td><td align="left" valign="top">0.21</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CamemBERT[cls]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.57</td><td align="left" valign="top">0.33</td><td align="left" valign="top"><italic>0.42</italic></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.38</td><td align="left" valign="top">0.28</td><td align="left" valign="top">0.32</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CamemBERT[entity]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.57</td><td align="left" valign="top">0.21</td><td align="left" valign="top">0.3</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.48</td><td align="left" valign="top">0.29</td><td align="left" valign="top"><italic>0.36</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CNN-based[cnnEncoder]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.71</td><td align="left" valign="top">0.04</td><td align="left" valign="top">0.07</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CNN-based[pcnnEncoder]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.1</td><td align="left" valign="top">0.71</td><td align="left" valign="top">0.17</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>For each event type, the highest <italic>F</italic><sub>1</sub>-score across input configurations is highlighted in italics format.</p></fn><fn id="table6fn2"><p><sup>b</sup>Configuration not evaluated.</p></fn></table-wrap-foot></table-wrap><p>Across all encoders, CamemBERT achieves the highest <italic>F</italic><sub>1</sub>-scores on the passage-based configurations (up to 0.42 for phenotypes and 0.58 for rare disease events), while ModernCamemBERT yields lower overall performance, with maximum <italic>F</italic><sub>1</sub>-scores of 0.27 for phenotypes and 0.39 for rare disease diagnoses. For transformer-based encoders, the [CLS]-based representation generally outperforms the entity-based strategy, particularly for CamemBERT. Results also show that standard CNN-based models achieve high precision (up to 0.75) but very low recall (0.04), while PCNN-based models improve recall (up to 0.83) at the expense of precision (0.1&#x2010;0.21).</p><p>According to the exact match evaluation (<xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>), our best LLM-based configurations outperform the OpenNRE-based models for both rare disease (0.73 vs 0.58 of <italic>F</italic><sub>1</sub>) and phenotype events (0.61 vs 0.42 of <italic>F</italic><sub>1</sub>).</p></sec><sec id="s3-3"><title>Application on the 2012 i2b2 Clinical Concepts</title><p><xref ref-type="table" rid="table7">Table 7</xref> reports exact-match and LLM-as-judge evaluation results for zero-shot and few-shot for TRE on the 2012 i2b2 clinical concepts using Qwen2.5, the best-performing model in our previous experiments.</p><table-wrap id="t7" position="float"><label>Table 7.</label><caption><p>Evaluation of zero-shot and few-shot relation extraction (i2b2 clinical concepts): precision, recall, and <italic>F</italic><sub>1</sub>-score.</p></caption><table id="table7" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom" colspan="6">Qwen2.5</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top" colspan="3">Exact match</td><td align="left" valign="top" colspan="3">LLM-as-a-judge<sup><xref ref-type="table-fn" rid="table7fn1">a</xref></sup></td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="7">Zero-shot extraction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text</td><td align="left" valign="top">0.25</td><td align="left" valign="top">0.39</td><td align="left" valign="top">0.31</td><td align="left" valign="top">0.27</td><td align="left" valign="top">0.43</td><td align="left" valign="top">0.34</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale</td><td align="left" valign="top">0.26</td><td align="left" valign="top">0.35</td><td align="left" valign="top">0.3</td><td align="left" valign="top">0.29</td><td align="left" valign="top">0.38</td><td align="left" valign="top">0.33</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+givenDates</td><td align="left" valign="top">0.31</td><td align="left" valign="top">0.46</td><td align="left" valign="top">0.37</td><td align="left" valign="top">0.31</td><td align="left" valign="top">0.47</td><td align="left" valign="top">0.38</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale+givenDates</td><td align="left" valign="top">0.35</td><td align="left" valign="top">0.47</td><td align="left" valign="top"><italic>0.4</italic><sup><xref ref-type="table-fn" rid="table7fn2">b</xref></sup></td><td align="left" valign="top">0.35</td><td align="left" valign="top">0.48</td><td align="left" valign="top"><italic>0.4</italic></td></tr><tr><td align="left" valign="top" colspan="7">Few-shot extraction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text</td><td align="left" valign="top">0.23</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.32</td><td align="left" valign="top">0.25</td><td align="left" valign="top">0.56</td><td align="left" valign="top">0.35</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale</td><td align="left" valign="top">0.28</td><td align="left" valign="top">0.46</td><td align="left" valign="top">0.35</td><td align="left" valign="top">0.29</td><td align="left" valign="top">0.49</td><td align="left" valign="top">0.36</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+givenDates</td><td align="left" valign="top">0.3</td><td align="left" valign="top">0.56</td><td align="left" valign="top">0.39</td><td align="left" valign="top">0.31</td><td align="left" valign="top">0.57</td><td align="left" valign="top"><italic>0.4</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale+givenDates</td><td align="left" valign="top">0.32</td><td align="left" valign="top">0.53</td><td align="left" valign="top"><italic>0.4</italic></td><td align="left" valign="top">0.32</td><td align="left" valign="top">0.54</td><td align="left" valign="top"><italic>0.4</italic></td></tr></tbody></table><table-wrap-foot><fn id="table7fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table7fn2"><p><sup>b</sup>For each evaluation, the highest <italic>F</italic><sub>1</sub>-score across input configurations is highlighted in italics format.</p></fn></table-wrap-foot></table-wrap><p>In the zero-shot setting, exact-match <italic>F</italic><sub>1</sub>-scores range from 0.30 to 0.40, with the best performance obtained with the Text+rationale+givenDates configuration. Similar results are observed under the LLM-as-a-judge evaluation, with a maximum <italic>F</italic><sub>1</sub> of 0.40. In the few-shot setting, performance further improves, with exact-match and LLM-as-a-judge <italic>F</italic><sub>1</sub>-scores reaching 0.40 for configurations that include the candidate dates, with or without rationale generation. Few-shot prompting consistently increases recall, reaching up to 0.57.</p></sec><sec id="s3-4"><title>End-to-End TRE and Normalization Results</title><p><xref ref-type="table" rid="table8">Table 8</xref> presents the exact-match evaluation of the end-to-end Qwen2.5 pipeline for TRE and normalization. The extraction component is evaluated under zero-shot and few-shot settings, whereas normalization is consistently performed in a few-shot configuration using 6 examples.</p><p>In the zero-shot settings, the highest <italic>F</italic><sub>1</sub>-scores are obtained when the model uses only the text as input (0.7 for rare disease and 0.62 for phenotypes). In the few-shot setting, performance increases to 0.72 for rare disease, whereas the phenotype score slightly decreases to 0.61, when rationale generation is included. These results indicate that few-shot examples enhance extraction quality for rare disease events while maintaining steady performance for phenotypes. Overall, performance remains stable across event types under the prompt-chaining configuration. The exact-match results for the remaining evaluated LLMs on the end-to-end TRE and normalization pipeline are presented in Tables S5 and S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><table-wrap id="t8" position="float"><label>Table 8.</label><caption><p>Exact-match evaluation of the end-to-end pipeline for temporal relation extraction and normalization.</p></caption><table id="table8" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom" colspan="6">Qwen2.5</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top" colspan="3">Rare disease</td><td align="left" valign="top" colspan="3">Phenotypes</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="7">Zero-shot extraction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.74</td><td align="left" valign="top">0.7</td><td align="left" valign="top">0.51</td><td align="left" valign="top">0.81</td><td align="left" valign="top"><italic>0.62</italic><sup><xref ref-type="table-fn" rid="table8fn1">a</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale</td><td align="left" valign="top">0.54</td><td align="left" valign="top">0.58</td><td align="left" valign="top">0.56</td><td align="left" valign="top">0.52</td><td align="left" valign="top">0.7</td><td align="left" valign="top">0.6</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+givenDates</td><td align="left" valign="top">0.59</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.63</td><td align="left" valign="top">0.46</td><td align="left" valign="top">0.71</td><td align="left" valign="top">0.56</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale+givenDates</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.5</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.58</td></tr><tr><td align="left" valign="top" colspan="7">Few-shot extraction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text</td><td align="left" valign="top">0.57</td><td align="left" valign="top">0.82</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.49</td><td align="left" valign="top">0.7</td><td align="left" valign="top">0.58</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale</td><td align="left" valign="top">0.72</td><td align="left" valign="top">0.73</td><td align="left" valign="top"><italic>0.72</italic></td><td align="left" valign="top">0.56</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.61</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+givenDates</td><td align="left" valign="top">0.52</td><td align="left" valign="top">0.7</td><td align="left" valign="top">0.6</td><td align="left" valign="top">0.38</td><td align="left" valign="top">0.59</td><td align="left" valign="top">0.46</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Text+rationale+givenDates</td><td align="left" valign="top">0.56</td><td align="left" valign="top">0.77</td><td align="left" valign="top">0.65</td><td align="left" valign="top">0.49</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.56</td></tr></tbody></table><table-wrap-foot><fn id="table8fn1"><p><sup>a</sup>For each event type, the highest <italic>F</italic><sub>1</sub>-score across input configurations is highlighted in italics format.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Environmental Impact</title><p>As shown in Table S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, temporal extraction and normalization inference experiments are estimated to have produced a total of around 21 kg of CO<sub>2</sub> equivalent, including around 5 kg CO<sub>2</sub> equivalent from temporal extraction and normalization experiments on the real-world corpus and more than 15 kg CO<sub>2</sub> equivalent from TRE on the i2b2 corpus using only Qwen2.5.</p><p>Among the evaluated LLMs, Llama3.3 had the highest environmental impact (2.3 kg of CO<sub>2</sub> equivalent), while Qwen2.5 achieved comparable performance on our clinical corpus with a lower carbon footprint (1.5 kg of CO<sub>2</sub> equivalent). Qwen3 and DeepSeek-R1 produced less than 1 kg of CO<sub>2</sub> equivalent.</p><p>A complete overview of the carbon emissions is provided in Table S8 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Prompting Strategies and Model Comparison in TRE</title><sec id="s4-1-1"><title>Zero-Shot Settings</title><p>As reported in <xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>, zero-shot prompting demonstrates clear differences in how models handle temporal reasoning using their pretrained knowledge only without prior contextual guidance.</p><p>Qwen2.5 consistently achieves the highest <italic>F</italic><sub>1</sub>-scores across both rare disease and phenotype events. Qwen2.5 and Llama3.3 display less variability in performance across configurations. The slight advantage of Qwen2.5 likely reflects the prompt refinement and example selection performed with this model during development, which enhanced alignment between the task formulation and its reasoning behavior. In contrast, the built-in reasoning models Qwen3 and DeepSeek-R1 show greater variability and lower precision. Their inherent reasoning capabilities occasionally lead to overextended or speculative date predictions, suggesting that unguided reasoning may compromise boundary precision when no contextual examples are provided. However, recall notably increases when candidate date lists are included (0.77 for Qwen3 and 0.92 for DeepSeek-R1 in rare disease extraction). This improvement suggests that explicit temporal cues help these reasoning models anchor their predictions.</p></sec><sec id="s4-1-2"><title>Few-Shot Setting</title><p>Few-shot prompting, even with a limited number of examples, consistently enhances overall relation extraction performance for rare disease diagnoses, though the magnitude of improvement varies considerably by model and input setting. As shown in <xref ref-type="table" rid="table2">Table 2</xref> for rare diseases, the gain induced by few-shot prompting makes Llama3.3 the top-performing model overall, with an <italic>F</italic><sub>1</sub> rising from 0.54 to 0.73 in the Text+rationale+givenDates configuration. DeepSeek-R1 and Qwen3 benefit the most for initially low-performing configurations (excluding rationale), while Qwen2.5 maintains near-saturated performance. However, phenotype events (<xref ref-type="table" rid="table3">Table 3</xref>) display inconsistent and limited gains across models and configurations, revealing fundamental task complexity differences.</p></sec><sec id="s4-1-3"><title>Impact of Textual Distance on Rare Disease and Phenotype Extraction</title><p>The impact of textual distance on extraction performance using Qwen2.5 is shown in <xref ref-type="fig" rid="figure4">Figures 4</xref> and <xref ref-type="fig" rid="figure5">5</xref>, with supplementary test data statistics provided in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Rare disease relation extraction demonstrates strong performance when events and temporal expressions are closely aligned (up to 0.73 at 14&#x2010;57 distance range). This could be due to disease mentions occurring early in the clinical narratives, within the initial presentation or history section, where they are closely aligned with nearby temporal expressions or the document date. <xref ref-type="fig" rid="figure4">Figure 4</xref> confirms this pattern, illustrating that with Qwen2.5, we observe better performance when rare disease mentions are located closer to their associated temporal expressions across all input configurations, and that extraction performance degrades when handling long-range dependencies (more than 315 characters). Additionally, each document generally contains fewer distinct rare disease entities, which reduces the relation extraction complexity and allows a strong overall performance with only 6 in-context examples. In contrast, as shown in <xref ref-type="fig" rid="figure5">Figure 5</xref>, phenotype relation extraction exhibits greater robustness across distance ranges, with <italic>F</italic><sub>1</sub> ranging from 0.52 to 0.82 for distances less than 1097 characters and from 0.20 to 0.85 within the 1097&#x2010;1999 distance range. This difference aligns with the underlying data distribution: rare disease relations are concentrated at short distances, while phenotype relations are more evenly distributed. Providing 14 in-context examples covering both short-range and long-distance temporal links during prompting allows the model to encounter diverse distance scenarios, likely supporting its relative stability across distance ranges. However, even when few-shot prompts include explicit instructions to identify temporal references, such as &#x201C;Report_date,&#x201D; extraction performance drops in the highest percentile bins. This indicates that while prompt design can partially mitigate the challenge, it cannot fully overcome the inherent difficulty of modeling long-range event-temporal dependencies.</p></sec><sec id="s4-1-4"><title>Impact of Input Configurations</title><p>Input configuration strongly affects extraction performance. Raw text alone tends to favor recall at the expense of precision, whereas adding rationale generation improves both precision and <italic>F</italic><sub>1</sub>. Providing candidate dates primarily increases recall but may reduce precision, as seen with DeepSeek-R1 in the zero-shot setting. Combining rationales with candidate dates produces the most balanced results, achieving the highest <italic>F</italic><sub>1</sub>-scores for rare diseases (0.70) and phenotypes (up to 0.61), underscoring the benefit of guided reasoning alongside temporal cues. These trends are generally supported by the CIs, with some variability across models and configurations.</p></sec><sec id="s4-1-5"><title>LLM-as-a-Judge Evaluation</title><p>Tables S3 and S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> report the extraction performance using an LLM-as-a-judge validation framework, allowing for minor temporal variations that strict exact-match metrics would penalize. Overall, this validation approach yields higher <italic>F</italic><sub>1</sub>-scores across models and input configurations, likely reflecting a more realistic assessment of semantically correct but format-divergent outputs, as confirmed by the strong agreement with the human judgment. Qwen2.5 achieved the highest <italic>F</italic><sub>1</sub>-scores in the zero-shot setting, reaching 0.74 for rare disease diagnoses and 0.62 for phenotypes. However, some errors persist in this evaluation process. For instance, a prediction of &#x201C;10/12/2020&#x201D; for a range &#x201C;10 au 12/12/2020&#x201D; may still be marked incorrect, while a relative expression such as &#x201C;ce jour&#x201D; (ie<italic>,</italic> &#x201C;that day&#x201D;) may be incorrectly validated when it refers to a previous consultation or hospitalization rather than the current document date. Despite these limitations, as noted earlier, a strong agreement with human judgment is observed when evaluating Qwen2.5 experiments, supporting the overall reliability of this evaluation approach.</p></sec></sec><sec id="s4-2"><title>Comparison With Rule-Based and Neural Baselines</title><p>We compare our LLM-based approach with a proximity-based rule baseline and supervised OpenNRE-based neural models. The proximity baseline achieves an <italic>F</italic><sub>1</sub> of 0.48 for rare disease diagnoses and 0.24 for phenotypes (<xref ref-type="table" rid="table5">Table 5</xref>), indicating that rare disease mentions are often temporally anchored by nearby expressions, whereas phenotype-temporal relations are more dispersed and varied, resulting in low recall. Several LLM configurations, including in zero-shot settings, substantially outperform the proximity baseline for phenotypes, with Qwen2.5 and Llama3.3 reaching <italic>F</italic><sub>1</sub>-scores of 0.55&#x2010;0.61 (gains of 31&#x2010;37 points). For rare disease relations, most zero-shot configurations also already outperform the proximity baseline, achieving up to 0.68 of <italic>F</italic><sub>1</sub> (Qwen2.5, Text+rationale+givenDates), and only 6 few-shot examples are sufficient to exceed 0.7 of <italic>F</italic><sub>1</sub>.</p><p>As reported in <xref ref-type="table" rid="table6">Table 6</xref>, OpenNRE-based models, trained jointly on a limited dataset of only 117 phenotype relations and 30 rare disease relations, show highly variable performance depending on architecture, input design, and pair selection strategy. CamemBERT generally outperforms ModernCamemBERT, and entity-based embeddings yield more balanced precision-recall than CLS-based representations. Passage-level inputs with LimitedPairs achieve the best <italic>F</italic><sub>1</sub> for rare disease relations (0.58), benefiting from local context and restricted candidate pairs, while FullText inputs improve recall for long-range links but reduce precision when allPairs are included. CNN-based encoders are inconsistent, often showing extreme precision-recall imbalances. Phenotype relations remain challenging, with the best neural configuration reaching an <italic>F</italic><sub>1</sub> of 0.42, underscoring the difficulty of capturing dispersed, long-distance temporal relations in clinical text, especially given the low amount of annotated training data.</p><p>These findings demonstrate the effectiveness of our LLM-based approach, which with few-shot examples substantially outperforms both neural and rule-based baselines for rare disease and phenotype relations with minimal annotation effort. While proximity-based rules offer interpretability, neural models typically require a substantial amount of annotated data, and both approaches struggle with long-distance, dispersed, and implicit relations. In contrast, our LLM-based approach robustly handles such complexities, including surface variations and temporal reasoning, providing clear advantages in low-resource clinical settings.</p></sec><sec id="s4-3"><title>Analysis of Relation Extraction on the i2b2 Clinical Concepts</title><p>As shown in <xref ref-type="table" rid="table7">Table 7</xref>, the evaluation of Qwen2.5 on the 2012 i2b2 clinical concepts demonstrates that our LLM-based approach generalizes effectively to a distinct clinical English dataset. While some studies [<xref ref-type="bibr" rid="ref53">53</xref>] explored the fine-tuning and prompt-tuning of LLaMA and GatorTron variants on the original i2b2 task, reporting competitive performance, a direct comparison with our experiments is not straightforward since we adapt the corpus to our annotation scheme. These adaptations ensure consistency with our experimental framework but prevent a strict head-to-head comparison with prior LLM studies on the original corpus. Under this setting, Qwen2.5 achieves meaningful performance on 6008 annotated relations, with the best zero-shot configuration reaching an exact-match <italic>F</italic><sub>1</sub> of 0.4 and recall of 0.47, while few-shot prompting with only 10 examples improves recall to 0.53 while maintaining the same <italic>F</italic><sub>1</sub>. These results are further validated by the LLM-as-a-judge evaluation. This performance is noteworthy, given the inherent difficulty of the i2b2 corpus, which is characterized by dense temporal annotations, implicit relations, and relatively low interannotator agreement [<xref ref-type="bibr" rid="ref4">4</xref>]. Some errors of our LLM-based approach arise from the model&#x2019;s tendency to infer temporal relations from contextual cues, occasionally producing plausible but incorrect links, while others may reflect omissions or inconsistencies in the gold annotations.</p><p>Overall, these results confirm that our LLM-based approach offers a robust and scalable solution for clinical TRE, even on challenging benchmarks, while remaining generalizable across datasets and languages in low-resource settings.</p></sec><sec id="s4-4"><title>Human Evaluation and Error Analysis of the Qwen2.5 Relation Extraction</title><p>Evaluating LLMs for information extraction remains challenging because traditional metrics are limited not only in their ability to capture semantic equivalence but also in their capacity to account for contextual and annotation-related variability. LLMs frequently generate synonyms, abbreviations, paraphrases, or normalized temporal expressions that are clinically correct yet do not exactly match reference annotations [<xref ref-type="bibr" rid="ref70">70</xref>,<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref74">74</xref>]. We therefore conducted a human evaluation of Qwen2.5, selected as the best-performing model, to further assess the quality of its predictions beyond exact-match and LLM-as-a-judge evaluations. This human evaluation confirms that a substantial proportion of extracted temporal relations, initially judged as incorrect, is in fact correct upon contextual review. <xref ref-type="table" rid="table4">Table 4</xref> shows that Qwen2.5 achieves 0.71 of human-validated precision for few-shot phenotype relations and 0.85 for zero-shot rare disease relations, exceeding exact-match precision scores (<xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>). Overall, human validation recovers up to 32 points with respect to exact-match precision (0.85 vs 0.53, rare disease diagnoses), with gains ranging from 6 to 32 points across configurations.</p><p>Our qualitative analysis highlights several sources of error that account for the gap between automatic metrics and human judgments. A recurring issue is annotation incompleteness, which underrepresents the temporal information in clinical narratives. For instance, in &#x201C;crise d&#x2019;epilepsie en 2013&#x201D; (ie, epileptic seizure in 2013), the event and its temporal relation are annotated, but subsequent mentions like &#x201C;8 &#x00E0; 9 crises par an en 2015 &#x00E0; 3 crises par an en 2018&#x201D; (ie, 8 to 9 seizures per year in 2015 to 3 seizures per year in 2018) include coreferring event mentions (&#x201C;crises,&#x201D; ie, seizures) that are not annotated, leaving the corresponding temporal relations to &#x201C;2015&#x201D; and &#x201C;2018&#x201D; absent from the gold standard. Consequently, model predictions for these relations are penalized, despite being clinically valid. Similarly, for relations in &#x201C;rechute de la maladie en 2012&#x201D; (ie, relapse of the disease in 2012), the model is penalized because the referenced disease entity is not explicitly annotated as an event. These cases constitute annotation omissions rather than errors in the model&#x2019;s temporal reasoning.</p><p>Other mismatches arise from annotation conventions that could, for instance, prioritize certain temporal cues over others or separately annotate individual mentions of entities. In our scheme, report dates are annotated only when they correspond to clinically salient events (eg, consultation, hospitalization, or procedure dates), while document metadata are excluded. However, the model sometimes extracts nearby temporal references such as header dates (&#x201C;Paris, 18/02/2021&#x201D;) or validation timestamps (&#x201C;letter validated on 17/09/2019&#x201D;) but fall outside the annotation guidelines. Additionally, variant disease mentions such as &#x201C;oligo-arthrite juv&#x00E9;nile idiopathique&#x201D; (ie, Oligoarticular juvenile idiopathic arthritis) and &#x201C;AJI oligoarticulaire&#x201D; (ie, Oligoarticular JIA) denote the same underlying disease but are treated as distinct in the gold standard, penalizing correct temporal associations across synonymous nomenclature. When both &#x201C;AJI&#x201D; (ie, JIA) and &#x201C;arthrite juv&#x00E9;nile idiopathique&#x201D; (ie, juvenile idiopathic arthritis) are annotated separately, the model often assigns the same date to both mentions, but only one relation is counted as correct under exact-match scoring.</p><p>Another source of error arises from the intrinsic complexity of clinical text, including nonstandard formatting and ambiguous temporal scope. For instance, in &#x201C;Le 05/05/2003: Poids=42 kg ...&#x201D; (May 5, 2003: Weight =42 kg ...), the date should be associated exclusively with the weight measurement and not with subsequent events mentioned in the text. However, the model frequently incorrectly propagates such narrowly scoped dates to later events, resulting in incorrect temporal relations. Similarly, constructions such as &#x201C;depuis la derni&#x00E8;re hospitalization de jour: ...&#x201D; (since the last day hospitalization: ...) can lead to temporal references being extended beyond their intended scope. The model also struggles with relative temporal expressions such as &#x201C;ce jour&#x201D; (on this day) when contextual anchoring is ambiguous. In sections describing disease history, it sometimes assigns the report date to longitudinal events, even when explicitly instructed to return &#x201C;None&#x201D; if no temporal expression is present in such sections. The model also tends to favor selecting a concrete date over &#x201C;None&#x201D; when candidate lists include both options, resulting in false-positive temporal links.</p><p>Finally, some errors come from prompt specification gaps. While negated event mentions were excluded from the gold standard annotations, the model was not explicitly instructed to ignore them. Thus, it occasionally extracts temporal expressions linked to negated events, which are counted as errors under exact-match evaluation. Overall, our findings suggest that adapting annotation protocols and evaluation frameworks to better reflect the capabilities of LLMs, such as considering synonyms, coreferring events, and contextually valid temporal expressions, could provide more accurate assessments of model performance in clinical information extraction.</p></sec><sec id="s4-5"><title>Assessment of the End-to-End Prompt-Chaining Pipeline</title><p>The end-to-end prompt-chaining pipeline of Qwen2.5 integrates temporal expression extraction with subsequent normalization, illustrating the model&#x2019;s ability to handle sequential clinical information tasks. In this setup, input configurations affect the extraction step, while normalization is consistently guided by a small set of 6 in-context examples applied to the previously extracted expressions. As shown in <xref ref-type="table" rid="table8">Table 8</xref>, rare disease events achieve higher combined <italic>F</italic><sub>1</sub>-scores than phenotype events, suggesting that extraction is the primary driver of overall pipeline performance. The stability of results across input configurations indicates that prompt chaining effectively preserves dependencies between extraction and normalization, limiting error propagation. Even with a small number of in-context examples for normalization, the model can reliably standardize temporal expressions, highlighting that improvements in overall performance are mostly attributable to extraction quality. Most normalization errors involved partial dates such as &#x201C;July&#x201D; being normalized as &#x201C;0001-07-01&#x201D; instead of &#x201C;2021-07-01,&#x201D; due to missing guidance for such cases and the lack of access to the surrounding clinical context during normalization. Conversely, some cases showed strong normalization despite extraction mismatches, when the model directly produced a normalized date rather than its original textual form, which is counted as incorrect for extraction but correct for normalization.</p><p>Overall, the stability of results across configurations indicates that the prompt-chaining approach effectively maintains dependencies between extraction and normalization tasks.</p></sec><sec id="s4-6"><title>Carbon Footprint</title><p>The estimation of carbon emissions of our inference experiments highlights a substantial variability across models, corpora, and prompting strategies. As summarized in Table S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, the TRE experiments on the i2b2 corpus using Qwen2.5 generated the highest carbon footprint. This is mainly due to the scale of the corpus, which contains 120 documents and 10,087 relations (<xref ref-type="table" rid="table1">Table 1</xref>), compared to our clinical corpus, which contains 42 documents with 356 phenotype relations and 70 rare disease relations. Qwen3 and DeepSeek-R1 produced lower CO<sub>2</sub> equivalent (less than 1 kg), which could be due to the limitation of generated tokens we fixed to 2000 to prevent overly long reasoning of these models. As illustrated in Table S8 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, few-shot prompting consistently increased carbon emissions compared with zero-shot prompting. The addition of rationale generation or given the list of possible dates seems to amplify costs due to longer input contexts and generated outputs.</p></sec><sec id="s4-7"><title>Limitations</title><p>The limited availability of annotated clinical data, which defines the low-resource settings targeted in our study, results in a relatively small evaluation dataset. Future work on larger annotated datasets would help further validate and generalize our approach.</p><p>Performance in temporal extraction is partly constrained by the dataset and task design. In our annotated corpus, nearly all temporal expressions associated with phenotypes or rare disease diagnoses are dates, with very few cases involving durations and almost none involving frequencies, limiting the assessment of model generalization to more diverse temporal phenomena. The fixed question formulation used for prompting further restricts coverage of complex or implicit temporal relations in clinical narratives, and annotation omissions or inconsistencies can penalize clinically valid predictions.</p><p>We evaluated multiple open-weight LLMs, but detailed qualitative error analysis and human validation were conducted only for the best-performing Qwen2.5. Additional experiments with smaller or more lightweight models may be needed, as they could offer lower computational cost for real-world deployment. We deliberately chose a zero- and few-shot prompting approach over fine-tuning to prioritize low-resource and easily deployable settings.</p><p>While the task of identifying and standardizing the date of an event may seem common, finding points of comparison with previous work remains difficult. We have tried to adapt the i2b2 corpus, which is widely used in the literature, to this task, but these adaptations do not allow for comparison with previous systems.</p><p>Carbon footprint measurements cover inference experiments only, excluding prompt refinement. These measurements were initially difficult to obtain reliably in the shared Ollama environment due to request queuing and dynamic model loading, though efficiency is a critical factor for clinical deployment.</p></sec><sec id="s4-8"><title>Conclusions</title><p>In this paper, we evaluated 4 open-weight, on-premises LLMs for TRE from French clinical narratives in a low-resource setting. We framed the task as question answering, combining zero- and few-shot prompting with prompt chaining for temporal normalization. Using a constructed and annotated real-world French clinical corpus, we compared model performance against rule-based and neural baselines through exact-match metrics, LLM-as-a-judge evaluation, and human validation. Results show that strong performance is achievable with few in-context examples for both phenotypes and rare disease diagnoses. Human validation of the best-performing Qwen2.5 predictions demonstrates that clinically correct temporal relations are often underestimated by traditional exact-match evaluations. We further demonstrate the generalizability of our approach on the English 2012 i2b2 corpus. Overall, carefully prompted generative LLMs can deliver high-quality TRE and normalization. Future work should explore annotation and evaluation frameworks better aligned with the strengths of LLMs to enable more accurate and clinically meaningful assessments.</p></sec></sec></body><back><notes><sec><title>Funding</title><p>This work was supported by state funding from The French National Research Agency (ANR) under the C&#x2019;IL-LICO project (ANR-17-RHUS-0002) and CDE.AI (ANR-21-PMRB-0002).</p></sec><sec><title>Data Availability</title><p>The deidentified dataset used in this study is not publicly available due to the sensitive nature of the text data and privacy concerns. The 2012 i2b2 dataset is available for research purposes under a Data Use Agreement.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: NB, XT, MV</p><p>Methodology: NB, XT, MV</p><p>Data curation: NB, GA, ASJ</p><p>Investigation: NB</p><p>Software: NB</p><p>Formal analysis: NB, XT, MV</p><p>Visualization: NB</p><p>Supervision: XT, MV</p><p>Validation: XT, MV</p><p>Resources: OB</p><p>Writing&#x2014;original draft: NB</p><p>Writing&#x2014;review and editing: NB, XT, MV, GA, ASJ, NG</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb2">TRE</term><def><p>temporal relation extraction</p></def></def-item><def-item><term id="abb3">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb4">CNN</term><def><p>convolutional neural network</p></def></def-item><def-item><term id="abb5">PCNN</term><def><p>piecewise convolutional neural network</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Escudi&#x00E9;</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Rance</surname><given-names>B</given-names> </name><name name-style="western"><surname>Malamut</surname><given-names>G</given-names> </name><etal/></person-group><article-title>A novel data-driven workflow combining literature and electronic health records to estimate comorbidities burden for a specific disease: a case study on autoimmune comorbidities in patients with celiac disease</article-title><source>BMC Med Inform Decis Mak</source><year>2017</year><month>09</month><day>29</day><volume>17</volume><issue>1</issue><fpage>140</fpage><pub-id pub-id-type="doi">10.1186/s12911-017-0537-y</pub-id><pub-id pub-id-type="medline">28962565</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>D</given-names> </name><name name-style="western"><surname>He</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Clinical concept extraction: a methodology review</article-title><source>J Biomed Inform</source><year>2020</year><month>09</month><volume>109</volume><fpage>103526</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2020.103526</pub-id><pub-id pub-id-type="medline">32768446</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jouffroy</surname><given-names>J</given-names> </name><name name-style="western"><surname>Feldman</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Lerner</surname><given-names>I</given-names> </name><name name-style="western"><surname>Rance</surname><given-names>B</given-names> </name><name name-style="western"><surname>Burgun</surname><given-names>A</given-names> </name><name name-style="western"><surname>Neuraz</surname><given-names>A</given-names> </name></person-group><article-title>Hybrid deep learning for medication-related information extraction from clinical texts in French: MedExt algorithm development study</article-title><source>JMIR Med Inform</source><year>2021</year><month>03</month><day>16</day><volume>9</volume><issue>3</issue><fpage>e17934</fpage><pub-id pub-id-type="doi">10.2196/17934</pub-id><pub-id pub-id-type="medline">33724196</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>W</given-names> </name><name name-style="western"><surname>Rumshisky</surname><given-names>A</given-names> </name><name name-style="western"><surname>Uzuner</surname><given-names>O</given-names> </name></person-group><article-title>Evaluating temporal relations in clinical text: 2012 i2b2 Challenge</article-title><source>J Am Med Inform Assoc</source><year>2013</year><volume>20</volume><issue>5</issue><fpage>806</fpage><lpage>813</lpage><pub-id pub-id-type="doi">10.1136/amiajnl-2013-001628</pub-id><pub-id pub-id-type="medline">23564629</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Derczynski</surname><given-names>L</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>G</given-names> </name><name name-style="western"><surname>Pustejovsky</surname><given-names>J</given-names> </name><name name-style="western"><surname>Verhagen</surname><given-names>M</given-names> </name></person-group><article-title>SemEval-2015 Task 6: Clinical TempEval</article-title><conf-name>Proceedings of the 9th International Workshop on Semantic Evaluation (SemEval 2015)</conf-name><conf-date>Jun 4-5, 2015</conf-date><pub-id pub-id-type="doi">10.18653/v1/S15-2136</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>WT</given-names> </name><name name-style="western"><surname>Derczynski</surname><given-names>L</given-names> </name><name name-style="western"><surname>Pustejovsky</surname><given-names>J</given-names> </name><name name-style="western"><surname>Verhagen</surname><given-names>M</given-names> </name></person-group><article-title>Task 12: clinical TempEval</article-title><conf-name>Proceedings of the 10th International Workshop on Semantic Evaluation (SemEval-2016)</conf-name><conf-date>Jun 16-17, 2016</conf-date><pub-id pub-id-type="doi">10.18653/v1/S16-1165</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>G</given-names> </name><name name-style="western"><surname>Palmer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Pustejovsky</surname><given-names>J</given-names> </name></person-group><article-title>Task 12: clinical TempEval</article-title><conf-name>Proceedings of the 11th International Workshop on Semantic Evaluation (SemEval-2017)</conf-name><conf-date>Aug 3-4, 2017</conf-date><pub-id pub-id-type="doi">10.18653/v1/S17-2093</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chang</surname><given-names>YC</given-names> </name><name name-style="western"><surname>Dai</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>JCY</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Tsai</surname><given-names>RTH</given-names> </name><name name-style="western"><surname>Hsu</surname><given-names>WL</given-names> </name></person-group><article-title>TEMPTING system: a hybrid method of rule and machine learning for temporal relation extraction in patient discharge summaries</article-title><source>J Biomed Inform</source><year>2013</year><month>12</month><volume>46 Suppl</volume><fpage>S54</fpage><lpage>S62</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2013.09.007</pub-id><pub-id pub-id-type="medline">24060600</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Kreimeyer</surname><given-names>K</given-names> </name><name name-style="western"><surname>Woo</surname><given-names>EJ</given-names> </name><etal/></person-group><article-title>A new algorithmic approach for the extraction of temporal associations from clinical narratives with an application to medical product safety surveillance reports</article-title><source>J Biomed Inform</source><year>2016</year><month>08</month><volume>62</volume><fpage>78</fpage><lpage>89</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2016.06.006</pub-id><pub-id pub-id-type="medline">27327528</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Najafabadipour</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zanin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rodr&#x00ED;guez-Gonz&#x00E1;lez</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Reconstructing the patient&#x2019;s natural history from electronic health records</article-title><source>Artif Intell Med</source><year>2020</year><month>05</month><volume>105</volume><fpage>101860</fpage><pub-id pub-id-type="doi">10.1016/j.artmed.2020.101860</pub-id><pub-id pub-id-type="medline">32505419</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>C</given-names> </name><name name-style="western"><surname>Dligach</surname><given-names>D</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>TA</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>GK</given-names> </name></person-group><article-title>Multilayered temporal modeling for the clinical domain</article-title><source>J Am Med Inform Assoc</source><year>2016</year><month>03</month><volume>23</volume><issue>2</issue><fpage>387</fpage><lpage>395</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocv113</pub-id><pub-id pub-id-type="medline">26521301</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Tourille</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ferret</surname><given-names>O</given-names> </name><name name-style="western"><surname>N&#x00E9;v&#x00E9;ol</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tannier</surname><given-names>X</given-names> </name></person-group><article-title>LIMSI-COT at SemEval-2016 Task 12: temporal relation identification using a pipeline of classifiers</article-title><conf-name>Proceedings of the 10th International Workshop on Semantic Evaluation (SemEval-2016)</conf-name><conf-date>Jun 16-17, 2016</conf-date><pub-id pub-id-type="doi">10.18653/v1/S16-1175</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Viani</surname><given-names>N</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>TA</given-names> </name><name name-style="western"><surname>Napolitano</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Supervised methods to extract clinical events from cardiology reports in Italian</article-title><source>J Biomed Inform</source><year>2019</year><month>07</month><volume>95</volume><fpage>103219</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2019.103219</pub-id><pub-id pub-id-type="medline">31150777</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alfattni</surname><given-names>G</given-names> </name><name name-style="western"><surname>Peek</surname><given-names>N</given-names> </name><name name-style="western"><surname>Nenadic</surname><given-names>G</given-names> </name></person-group><article-title>Attention-based bidirectional long short-term memory networks for extracting temporal relationships from clinical discharge summaries</article-title><source>J Biomed Inform</source><year>2021</year><month>11</month><volume>123</volume><fpage>103915</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2021.103915</pub-id><pub-id pub-id-type="medline">34600144</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Olex</surname><given-names>AL</given-names> </name><name name-style="western"><surname>McInnes</surname><given-names>BT</given-names> </name></person-group><article-title>Review of temporal reasoning in the clinical domain for timeline extraction: where we are and where we need to be</article-title><source>J Biomed Inform</source><year>2021</year><month>06</month><volume>118</volume><fpage>103784</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2021.103784</pub-id><pub-id pub-id-type="medline">33862232</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>X</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>T</given-names> </name><name name-style="western"><surname>Yao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ye</surname><given-names>D</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>M</given-names> </name></person-group><article-title>OpenNRE: an open and extensible toolkit for neural relation extraction</article-title><conf-name>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</conf-name><conf-date>Nov 3-7, 2019</conf-date><pub-id pub-id-type="doi">10.18653/v1/D19-3029</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Tourille</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ferret</surname><given-names>O</given-names> </name><name name-style="western"><surname>Tannier</surname><given-names>X</given-names> </name><name name-style="western"><surname>N&#x00E9;v&#x00E9;ol</surname><given-names>A</given-names> </name></person-group><article-title>LIMSI-COT at SemEval-2017 Task 12: neural architecture for temporal information extraction from clinical narratives</article-title><conf-name>Proceedings of the 11th International Workshop on Semantic Evaluation (SemEval-2017)</conf-name><conf-date>Aug 3-4, 2017</conf-date><pub-id pub-id-type="doi">10.18653/v1/S17-2098</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Han</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Clinical temporal relation extraction with probabilistic soft logic regularization and global inference</article-title><source>AAAI</source><year>2021</year><volume>35</volume><issue>16</issue><fpage>14647</fpage><lpage>14655</lpage><pub-id pub-id-type="doi">10.1609/aaai.v35i16.17721</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dligach</surname><given-names>D</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>C</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>G</given-names> </name></person-group><article-title>Neural temporal relation extraction</article-title><conf-name>Proceedings of the 15th Conference of the European Chapter of the Association for Computational Linguistics: Volume 2, Short Papers</conf-name><conf-date>Apr 3-7, 2017</conf-date><pub-id pub-id-type="doi">10.18653/v1/E17-2118</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Gaizauskas</surname><given-names>R</given-names> </name><name name-style="western"><surname>Harkema</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hepple</surname><given-names>M</given-names> </name><name name-style="western"><surname>Setzer</surname><given-names>A</given-names> </name></person-group><article-title>Task-oriented extraction of temporal information: the case of clinical narratives</article-title><conf-name>Proceedings of the International Workshop on Temporal Representation and Reasoning</conf-name><conf-date>Jun 15-17, 2006</conf-date><pub-id pub-id-type="doi">10.1109/TIME.2006.27</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chikka</surname><given-names>VR</given-names> </name></person-group><article-title>CDE-IIITH at SemEval-2016 Task 12: extraction of temporal information from clinical documents using machine learning techniques</article-title><conf-name>Proceedings of the 10th International Workshop on Semantic Evaluation (SemEval-2016)</conf-name><conf-date>Jun 16-17, 2016</conf-date><pub-id pub-id-type="doi">10.18653/v1/S16-1192</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>P</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>H</given-names> </name></person-group><article-title>UTA DLNLP at SemEval-2016 Task 12: deep learning based natural language processing system for clinical information identification from clinical notes and pathology reports</article-title><conf-name>Proceedings of the 10th International Workshop on Semantic Evaluation (SemEval-2016)</conf-name><conf-date>Jun 16-17, 2016</conf-date><pub-id pub-id-type="doi">10.18653/v1/S16-1197</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Tourille</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ferret</surname><given-names>O</given-names> </name><name name-style="western"><surname>Tannier</surname><given-names>X</given-names> </name><name name-style="western"><surname>Neveol</surname><given-names>A</given-names> </name></person-group><article-title>Temporal information extraction from clinical text</article-title><conf-name>Proceedings of the 15th Conference of the European Chapter of the Association for Computational Linguistics: Volume 2, Short Papers</conf-name><conf-date>Apr 3-7, 2017</conf-date><pub-id pub-id-type="doi">10.18653/v1/E17-2117</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Galvan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Okazaki</surname><given-names>N</given-names> </name><name name-style="western"><surname>Matsuda</surname><given-names>K</given-names> </name><name name-style="western"><surname>Inui</surname><given-names>K</given-names> </name></person-group><article-title>Investigating the challenges of temporal relation extraction from clinical text</article-title><conf-name>Proceedings of the Ninth International Workshop on Health Text Mining and Information Analysis</conf-name><conf-date>Oct 31, 2018</conf-date><pub-id pub-id-type="doi">10.18653/v1/W18-5607</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wolf</surname><given-names>T</given-names> </name><name name-style="western"><surname>Debut</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sanh</surname><given-names>V</given-names> </name></person-group><article-title>Transformers: state-of-the-art natural language processing</article-title><conf-name>Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 16-20, 2020</conf-date><pub-id pub-id-type="doi">10.18653/v1/2020.emnlp-demos.6</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>C</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>T</given-names> </name><name name-style="western"><surname>Dligach</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>G</given-names> </name></person-group><article-title>A BERT-based universal model for both within- and cross-sentence clinical temporal relation extraction</article-title><conf-name>Proceedings of the 2nd Clinical Natural Language Processing Workshop</conf-name><conf-date>Jun 7, 2019</conf-date><pub-id pub-id-type="doi">10.18653/v1/W19-1908</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Miller</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Dligach</surname><given-names>D</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>G</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Demner-fushman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>K</given-names> </name></person-group><article-title>End-to-end clinical temporal information extraction with multi-head attention</article-title><conf-name>The 22nd Workshop on Biomedical Natural Language Processing and BioNLP Shared Tasks</conf-name><conf-date>Jul 13, 2023</conf-date><pub-id pub-id-type="doi">10.18653/v1/2023.bionlp-1.28</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Styler</surname><given-names>WF</given-names>  <suffix>4th</suffix></name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Finan</surname><given-names>S</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Lin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Collins</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>L</given-names> </name></person-group><article-title>Temporal annotation in the clinical domain</article-title><source>Trans Assoc Comput Linguist</source><year>2014</year><month>04</month><volume>2</volume><fpage>143</fpage><lpage>154</lpage><pub-id pub-id-type="doi">10.1162/tacl_a_00172</pub-id><pub-id pub-id-type="medline">29082229</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zeng</surname><given-names>D</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>K</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>J</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>M&#x00E0;rquez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Callison-Burch</surname><given-names>C</given-names> </name><name name-style="western"><surname>Su</surname><given-names>J</given-names> </name></person-group><article-title>Distant supervision for relation extraction via piecewise convolutional neural networks</article-title><conf-name>Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Sep 17-21, 2015</conf-date><pub-id pub-id-type="doi">10.18653/v1/D15-1203</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chaturvedi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Baghershahi</surname><given-names>P</given-names> </name><name name-style="western"><surname>Medya</surname><given-names>S</given-names> </name><name name-style="western"><surname>Di Eugenio</surname><given-names>B</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Che</surname><given-names>W</given-names> </name><name name-style="western"><surname>Nabende</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shutova</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pilehvar</surname><given-names>MT</given-names> </name></person-group><article-title>Temporal relation extraction in clinical texts: a span-based graph transformer approach</article-title><conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.1251</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ning</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Roth</surname><given-names>D</given-names> </name></person-group><article-title>A multi-axis annotation scheme for event temporal relations</article-title><conf-name>Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</conf-name><conf-date>Jul 15-20, 2018</conf-date><pub-id pub-id-type="doi">10.18653/v1/P18-1122</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bannour</surname><given-names>N</given-names> </name><name name-style="western"><surname>Rance</surname><given-names>B</given-names> </name><name name-style="western"><surname>Tannier</surname><given-names>X</given-names> </name><name name-style="western"><surname>Neveol</surname><given-names>A</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Demner-fushman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>K</given-names> </name></person-group><article-title>Event-independent temporal positioning: application to french clinical text</article-title><conf-name>The 22nd Workshop on Biomedical Natural Language Processing and BioNLP Shared Tasks</conf-name><conf-date>Jul 13, 2023</conf-date><pub-id pub-id-type="doi">10.18653/v1/2023.bionlp-1.16</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Andrew</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Potier</surname><given-names>J</given-names> </name><name name-style="western"><surname>Garcelon</surname><given-names>N</given-names> </name><name name-style="western"><surname>Burgun</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vincent</surname><given-names>M</given-names> </name></person-group><article-title>Using large language models for temporal relation extraction from pediatric clinical reports</article-title><source>JAMIA Open</source><year>2025</year><month>12</month><volume>8</volume><issue>6</issue><fpage>ooaf121</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooaf121</pub-id><pub-id pub-id-type="medline">41281245</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gumiel</surname><given-names>YB</given-names> </name><name name-style="western"><surname>Silva e Oliveira</surname><given-names>LE</given-names> </name><name name-style="western"><surname>Claveau</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Temporal relation extraction in clinical texts</article-title><source>ACM Comput Surv</source><year>2022</year><month>09</month><day>30</day><volume>54</volume><issue>7</issue><fpage>1</fpage><lpage>36</lpage><pub-id pub-id-type="doi">10.1145/3462475</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>M</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Goyal</surname><given-names>N</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Jurafsky</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schluter</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tetreault</surname><given-names>J</given-names> </name></person-group><article-title>BART: denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension</article-title><conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>Jul 5-10, 2020</conf-date><pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.703</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raffel</surname><given-names>C</given-names> </name><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Exploring the limits of transfer learning with a unified text-to-text transformer</article-title><source>J Mach Learn Res</source><year>2020</year><access-date>2026-09-13</access-date><volume>21</volume><issue>1</issue><comment><ext-link ext-link-type="uri" xlink:href="https://jmlr.org/papers/v21/20-074.html">https://jmlr.org/papers/v21/20-074.html</ext-link></comment></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Radford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Narasimhan</surname><given-names>K</given-names> </name></person-group><article-title>Improving language understanding by generative pre-training</article-title><source>OpenAI</source><access-date>2026-06-13</access-date><comment>Preprint posted online on 2018</comment><comment><ext-link ext-link-type="uri" xlink:href="https://cdn.openai.com/research-covers/language-unsupervised/language_understanding_paper.pdf">https://cdn.openai.com/research-covers/language-unsupervised/language_understanding_paper.pdf</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Touvron</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lavril</surname><given-names>T</given-names> </name><name name-style="western"><surname>Izacard</surname><given-names>G</given-names> </name><etal/></person-group><article-title>LLaMA: open and efficient foundation language models</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 27, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2302.13971</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dligach</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>T</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>G</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>K</given-names> </name><name name-style="western"><surname>Rumshisky</surname><given-names>A</given-names> </name></person-group><article-title>Exploring text representations for generative temporal relation extraction</article-title><conf-name>Proceedings of the 4th Clinical Natural Language Processing Workshop</conf-name><conf-date>Jul 14, 2022</conf-date><pub-id pub-id-type="doi">10.18653/v1/2022.clinicalnlp-1.12</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Saiz</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Altuna</surname><given-names>B</given-names> </name></person-group><article-title>End-to-end temporal relation extraction in the clinical domain</article-title><access-date>2026-09-13</access-date><conf-name>Proceedings of the Text2Story&#x2019;23 Workshop</conf-name><conf-date>Apr 2, 2023</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://text2story23.inesctec.pt/">https://text2story23.inesctec.pt/</ext-link></comment></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Huguet Cabot</surname><given-names>PL</given-names> </name><name name-style="western"><surname>Navigli</surname><given-names>R</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Moens</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Specia</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tau</surname><given-names>YS</given-names> </name></person-group><article-title>REBEL: relation extraction by end-to-end language generation</article-title><conf-name>Findings of the Association for Computational Linguistics</conf-name><conf-date>Nov 7-11, 2021</conf-date><pub-id pub-id-type="doi">10.18653/v1/2021.findings-emnlp.204</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yuan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Demner-fushman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>K</given-names> </name></person-group><article-title>Zero-shot temporal relation extraction with ChatGPT</article-title><conf-name>The 22nd Workshop on Biomedical Natural Language Processing and BioNLP Shared Tasks</conf-name><conf-date>Jul 2023</conf-date><pub-id pub-id-type="doi">10.18653/v1/2023.bionlp-1.7</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Tang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Han</surname><given-names>X</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>X</given-names> </name></person-group><article-title>Does synthetic data generation of LLMs help clinical text mining</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 10, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.04360</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Su</surname><given-names>X</given-names> </name><name name-style="western"><surname>Howard</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name></person-group><article-title>Transformer-based temporal information extraction and application: a review</article-title><conf-name>Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 4-9, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.emnlp-main.1467</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning</article-title><source>Nature</source><year>2025</year><month>09</month><volume>645</volume><issue>8081</issue><fpage>633</fpage><lpage>638</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-09422-z</pub-id><pub-id pub-id-type="medline">40962978</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Li</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Qwen3 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  May 14, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.09388</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kougia</surname><given-names>V</given-names> </name><name name-style="western"><surname>Sedova</surname><given-names>A</given-names> </name><name name-style="western"><surname>Stephan</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Zaporojets</surname><given-names>K</given-names> </name><name name-style="western"><surname>Roth</surname><given-names>B</given-names> </name></person-group><article-title>Analysing zero-shot temporal relation extraction on clinical notes using temporal consistency</article-title><conf-name>Proceedings of the 23rd Workshop on Biomedical Natural Language Processing</conf-name><conf-date>Aug 16, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.bionlp-1.6</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bannour</surname><given-names>N</given-names> </name><name name-style="western"><surname>Andrew</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Vincent</surname><given-names>M</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ben Abacha</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bitterman</surname><given-names>D</given-names> </name></person-group><article-title>Team NLPeers at Chemotimelines 2024: evaluation of two timeline extraction methods, can generative LLM do it all or is smaller model fine-tuning still relevant?</article-title><conf-name>Proceedings of the 6th Clinical Natural Language Processing Workshop</conf-name><conf-date>Jun 21, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.clinicalnlp-1.39</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hochheiser</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>WJ</given-names> </name><name name-style="western"><surname>Goldner</surname><given-names>E</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>G</given-names> </name></person-group><article-title>Overview of the 2024 shared task on chemotherapy treatment timeline extraction</article-title><conf-name>Proceedings of the 6th Clinical Natural Language Processing Workshop</conf-name><conf-date>Jun 21, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.clinicalnlp-1.53</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peng</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>KE</given-names> </name><etal/></person-group><article-title>Model tuning or prompt tuning? A study of large language models for clinical concept and relation extraction</article-title><source>J Biomed Inform</source><year>2024</year><month>05</month><volume>153</volume><fpage>104630</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2024.104630</pub-id><pub-id pub-id-type="medline">38548007</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peng</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Generative large language models are all-purpose text analytics engines: text-to-text learning is all your need</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>09</month><day>1</day><volume>31</volume><issue>9</issue><fpage>1892</fpage><lpage>1903</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae078</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yeh</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Lavergne</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zweigenbaum</surname><given-names>P</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Calzolari</surname><given-names>N</given-names> </name><name name-style="western"><surname>B&#x00E9;chet</surname><given-names>F</given-names> </name><name name-style="western"><surname>Blache</surname><given-names>P</given-names> </name></person-group><article-title>Decorate the examples: a simple method of prompt design for biomedical relation extraction</article-title><conf-name>Thirteenth Language Resources and Evaluation Conference</conf-name><pub-id pub-id-type="doi">10.63317/2x38wgwt3eai</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rasmy</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Prompting large language models for clinical temporal relation extraction</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 4, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.04512</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Haddadan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Le</surname><given-names>TD</given-names> </name><name name-style="western"><surname>Duong</surname><given-names>T</given-names> </name><name name-style="western"><surname>Thieu</surname><given-names>T</given-names> </name></person-group><article-title>LAILab at Chemotimelines 2024: finetuning sequence-to-sequence language models for temporal relation extraction towards cancer patient undergoing chemotherapy treatment</article-title><conf-name>Proceedings of the 6th Clinical Natural Language Processing Workshop</conf-name><conf-date>Jun 21, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.clinicalnlp-1.37</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chung</surname><given-names>HW</given-names> </name><name name-style="western"><surname>Hou</surname><given-names>L</given-names> </name><name name-style="western"><surname>Longpre</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Scaling instruction-finetuned language models</article-title><source>J Mach Learn Res</source><year>2024</year><access-date>2026-09-13</access-date><volume>25</volume><issue>70</issue><fpage>1</fpage><lpage>53</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://jmlr.org/papers/v25/23-0870.html">https://jmlr.org/papers/v25/23-0870.html</ext-link></comment></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hochheiser</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>W</given-names> </name><name name-style="western"><surname>Goldner</surname><given-names>ET</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>GK</given-names> </name></person-group><article-title>Overview of the 2025 shared task on chemotherapy treatment timeline extraction</article-title><access-date>2026-02-02</access-date><conf-name>Proceedings of the 7th Clinical Natural Language Processing Workshop</conf-name><conf-date>Oct 30, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.clinicalnlp-1.1/">https://aclanthology.org/2025.clinicalnlp-1.1/</ext-link></comment></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Vydiswaran</surname><given-names>VGV</given-names> </name></person-group><article-title>Team NLP4Health at ChemoTimelines 2025: finetuning large language models for temporal relation extractions from clinical notes</article-title><access-date>2026-02-02</access-date><conf-name>Proceedings of the 7th Clinical Natural Language Processing Workshop</conf-name><conf-date>Oct 30, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.clinicalnlp-1.4/">https://aclanthology.org/2025.clinicalnlp-1.4/</ext-link></comment></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zaghir</surname><given-names>J</given-names> </name><name name-style="western"><surname>Naguib</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bjelogrlic</surname><given-names>M</given-names> </name><name name-style="western"><surname>N&#x00E9;v&#x00E9;ol</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tannier</surname><given-names>X</given-names> </name><name name-style="western"><surname>Lovis</surname><given-names>C</given-names> </name></person-group><article-title>Prompt engineering paradigms for medical applications: scoping review</article-title><source>J Med Internet Res</source><year>2024</year><month>09</month><day>10</day><volume>26</volume><fpage>e60501</fpage><pub-id pub-id-type="doi">10.2196/60501</pub-id><pub-id pub-id-type="medline">39255030</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mann</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ryder</surname><given-names>N</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Larochelle</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ranzato</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hadsell</surname><given-names>R</given-names> </name><name name-style="western"><surname>Balcan</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>H</given-names> </name></person-group><article-title>Language models are few-shot learners</article-title><access-date>2026-09-13</access-date><conf-name>Advances in Neural Information Processing Systems</conf-name><conf-date>Dec 6-12, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf">https://proceedings.neurips.cc/paper_files/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf</ext-link></comment></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garcelon</surname><given-names>N</given-names> </name><name name-style="western"><surname>Neuraz</surname><given-names>A</given-names> </name><name name-style="western"><surname>Salomon</surname><given-names>R</given-names> </name><etal/></person-group><article-title>A clinician friendly data warehouse oriented toward narrative reports: Dr. Warehouse</article-title><source>J Biomed Inform</source><year>2018</year><month>04</month><volume>80</volume><fpage>52</fpage><lpage>63</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2018.02.019</pub-id><pub-id pub-id-type="medline">29501921</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="web"><article-title>Orphanet nomenclature pack</article-title><source>ORPHAcodes</source><access-date>2026-02-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.orphacode.org/pack-nomenclature/">https://www.orphacode.org/pack-nomenclature/</ext-link></comment></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="web"><article-title>French language translation</article-title><source>Human Phenotype Ontology Internationalisation Effort</source><access-date>2026-02-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://obophenotype.github.io/hpo-translations/translations/fr/">https://obophenotype.github.io/hpo-translations/translations/fr/</ext-link></comment></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="web"><article-title>UMLS&#x00AE; 2023AA Release Available</article-title><source>National Library of Medicine</source><year>2023</year><access-date>2026-02-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.nlm.nih.gov/pubs/techbull/mj23/mj23_umls_2023aa_release.html">https://www.nlm.nih.gov/pubs/techbull/mj23/mj23_umls_2023aa_release.html</ext-link></comment></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Wajsburt</surname><given-names>P</given-names> </name><name name-style="western"><surname>Petit-Jean</surname><given-names>T</given-names> </name><name name-style="western"><surname>Dura</surname><given-names>B</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jean</surname><given-names>C</given-names> </name><name name-style="western"><surname>Bey</surname><given-names>R</given-names> </name></person-group><article-title>EDS-NLP: efficient information extraction from French clinical notes</article-title><source>Zenodo</source><access-date>2026-09-13</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/17273431">https://zenodo.org/records/17273431</ext-link></comment></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="web"><article-title>Dates</article-title><source>EDS-NLP</source><access-date>2026-02-03</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://aphp.github.io/edsnlp/latest/pipes/misc/dates/">https://aphp.github.io/edsnlp/latest/pipes/misc/dates/</ext-link></comment></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Stenetorp</surname><given-names>P</given-names> </name><name name-style="western"><surname>Pyysalo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Topi&#x0107;</surname><given-names>G</given-names> </name><name name-style="western"><surname>Ohta</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tsujii</surname><given-names>J</given-names> </name></person-group><article-title>BRAT: a web-based tool for NLP-assisted text annotation</article-title><conf-name>Proceedings of the Demonstrations Session at EACL 2012</conf-name><conf-date>Apr 23-27, 2012</conf-date><pub-id pub-id-type="doi">10.5555/2380921.2380942</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>W</given-names> </name><name name-style="western"><surname>Rumshisky</surname><given-names>A</given-names> </name><name name-style="western"><surname>Uzuner</surname><given-names>O</given-names> </name></person-group><article-title>Annotating temporal information in clinical narratives</article-title><source>J Biomed Inform</source><year>2013</year><month>12</month><volume>46</volume><fpage>S5</fpage><lpage>S12</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2013.07.004</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="web"><article-title>aphp/edsnlp at rule_based_relation</article-title><source>GitHub</source><access-date>2026-02-03</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/aphp/edsnlp/tree/rule_based_relation">https://github.com/aphp/edsnlp/tree/rule_based_relation</ext-link></comment></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Dekking</surname><given-names>FM</given-names> </name><name name-style="western"><surname>Kraaikamp</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lopuha&#x00E4;</surname><given-names>HP</given-names> </name><name name-style="western"><surname>Meester</surname><given-names>LE</given-names> </name></person-group><source>A Modern Introduction to Probability and Statistics</source><year>2005</year><publisher-name>Springer</publisher-name><pub-id pub-id-type="doi">10.1007/1-84628-168-7</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Fan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hou</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Calzolari</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kan</surname><given-names>MY</given-names> </name><name name-style="western"><surname>Hoste</surname><given-names>V</given-names> </name><name name-style="western"><surname>Lenci</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sakti</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xue</surname><given-names>N</given-names> </name></person-group><article-title>Evaluating generative language models in information extraction as subjective question correction</article-title><conf-name>Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)</conf-name><conf-date>May 20-25, 2024</conf-date><pub-id pub-id-type="doi">10.63317/2rbzu6ev8x4m</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A survey on evaluation of large language models</article-title><source>ACM Trans Intell Syst Technol</source><year>2024</year><month>06</month><day>30</day><volume>15</volume><issue>3</issue><fpage>1</fpage><lpage>45</lpage><pub-id pub-id-type="doi">10.1145/3641289</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="web"><source>gpt-oss:20b</source><access-date>2026-02-03</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://ollama.com/library/gpt-oss:20b">https://ollama.com/library/gpt-oss:20b</ext-link></comment></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lannelongue</surname><given-names>L</given-names> </name><name name-style="western"><surname>Grealey</surname><given-names>J</given-names> </name><name name-style="western"><surname>Inouye</surname><given-names>M</given-names> </name></person-group><article-title>Green Algorithms: quantifying the carbon footprint of computation</article-title><source>Adv Sci (Weinh)</source><year>2021</year><month>06</month><volume>8</volume><issue>12</issue><fpage>2100707</fpage><pub-id pub-id-type="doi">10.1002/advs.202100707</pub-id><pub-id pub-id-type="medline">34194954</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Laskar</surname><given-names>MTR</given-names> </name><name name-style="western"><surname>Jahan</surname><given-names>I</given-names> </name><name name-style="western"><surname>Dolatabadi</surname><given-names>E</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>C</given-names> </name><name name-style="western"><surname>Hoque</surname><given-names>E</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Che</surname><given-names>W</given-names> </name><name name-style="western"><surname>Nabende</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shutova</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pilehvar</surname><given-names>MT</given-names> </name></person-group><article-title>Improving automatic evaluation of large language models (LLMs) in biomedical relation extraction via LLMs-as-the-Judge</article-title><conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1)</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.1238</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Further details about data statistics and experimental results.</p><media xlink:href="jmir_v28i1e95198_app1.docx" xlink:title="DOCX File, 25 KB"/></supplementary-material></app-group></back></article>