<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMIR</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR AI</journal-id>
      <journal-title>JMIR AI</journal-title>
      <issn pub-type="epub">2817-1705</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v5i1e75561</article-id>
      <article-id pub-id-type="pmid">42467970</article-id>
      <article-id pub-id-type="doi">10.2196/75561</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Enhancing Large Language Models for Identifying and Prioritizing Important Medical Jargons From Electronic Health Record Notes Using Data Augmentation: Comparative Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Coristine</surname>
            <given-names>Andrew</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Wang</surname>
            <given-names>Yijun</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Du</surname>
            <given-names>Xinsong</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Manion</surname>
            <given-names>Frank</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" corresp="yes" equal-contrib="yes">
          <name name-style="western">
            <surname>Jang</surname>
            <given-names>Won Seok</given-names>
          </name>
          <degrees>MS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Miner School of Computer and Information Sciences</institution>
            <institution>University of Massachusetts Lowell</institution>
            <addr-line>1 University Avenue</addr-line>
            <addr-line>Lowell, MA, 01854</addr-line>
            <country>United States</country>
            <phone>1 9789344000</phone>
            <email>WonSeok_Jang@uml.edu</email>
          </address>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0001-5439-7299</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Sultana</surname>
            <given-names>Sharmin</given-names>
          </name>
          <degrees>BS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-8016-9329</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Yao</surname>
            <given-names>Zonghai</given-names>
          </name>
          <degrees>MS</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-5707-8410</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Tran</surname>
            <given-names>Hieu</given-names>
          </name>
          <degrees>MS</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0007-1035-7395</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Yang</surname>
            <given-names>Zhichao</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-2797-4257</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Kwon</surname>
            <given-names>Sunjae</given-names>
          </name>
          <degrees>MS</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-5425-6779</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author">
          <name name-style="western">
            <surname>Yu</surname>
            <given-names>Hong</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-9263-5035</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Miner School of Computer and Information Sciences</institution>
        <institution>University of Massachusetts Lowell</institution>
        <addr-line>Lowell, MA</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Manning College of Information &#38; Computer Sciences</institution>
        <institution>University of Massachusetts Amherst</institution>
        <addr-line>Amherst, MA</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <institution>Center for Healthcare Organization and Implementation Research</institution>
        <institution>VA Bedford Health Care</institution>
        <addr-line>Bedford, MA</addr-line>
        <country>United States</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Won Seok Jang <email>WonSeok_Jang@uml.edu</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>17</day>
        <month>7</month>
        <year>2026</year>
      </pub-date>
      <volume>5</volume>
      <elocation-id>e75561</elocation-id>
      <history>
        <date date-type="received">
          <day>6</day>
          <month>4</month>
          <year>2025</year>
        </date>
        <date date-type="rev-request">
          <day>22</day>
          <month>5</month>
          <year>2025</year>
        </date>
        <date date-type="accepted">
          <day>24</day>
          <month>4</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Won Seok Jang, Sharmin Sultana, Zonghai Yao, Hieu Tran, Zhichao Yang, Sunjae Kwon, Hong Yu. Originally published in JMIR AI (https://ai.jmir.org), 17.07.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on https://www.ai.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://ai.jmir.org/2026/1/e75561" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>OpenNotes allows patients to access their electronic health record (EHR) notes through online patient portals. However, EHR notes contain abundant medical jargon, which can be difficult for patients to comprehend. One way to improve comprehension is by reducing information overload and helping patients focus on the medical terms that matter most to them.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to evaluate both closed-source and open-source large language models (LLMs) for extracting and prioritizing medical jargon from EHR notes relevant to individual patients, leveraging prompting techniques, fine-tuning, and data augmentation.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>We evaluated the performance of closed-source and open-source LLMs on a dataset of 90 expert-annotated EHR notes. We tested various combinations of settings, including (1) general and structured prompts, (2) zero-shot and few-shot prompting, (3) fine-tuning, and (4) data augmentation. To enhance the extraction and prioritization capabilities of open-source models in low-resource settings, we applied data augmentation using GPT-4o and integrated a ranking technique to refine the training process. Additionally, to measure the impact of dataset size, we fine-tuned the models by incrementally increasing the size of the augmented dataset from 10 to 9995 and tested their performance. The effectiveness of the models was assessed using 10-fold cross-validation, providing a comprehensive evaluation across various settings. We report the <italic>F</italic><sub>1</sub>-score and mean reciprocal rank for performance evaluation using two different string matching algorithms (relaxed string matching and Jaccard Index). We also conducted an error analysis classifying the erroneous outputs from the models.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>Our results show that open-source models achieved the highest performance, particularly when using fine-tuning with a gold-standard dataset. Under Jaccard Index–based string matching, DeepSeek 8B set the benchmarks with an <italic>F</italic><sub>1</sub>-score of 0.431 (SD 0.046); similarly, BioMistral 7B showed a mean reciprocal rank of 0.577 (SD 0.109). However, under relaxed string matching, open-source models were unable to match the performance of closed-source models, even with data augmentation or fine-tuning. We analyzed our experiment from several perspectives. First, few-shot prompting did not show an advantage over zero-shot prompting in vanilla models. Second, when comparing general and structured prompts, we found that model performance could deviate largely based on prompting styles. Third, fine-tuning with a small gold-standard dataset improved performance. Finally, data augmentation yielded performance comparable to or even surpassing the fine-tuning strategy. However, it also underscored the importance of the quality of the augmented dataset.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>The evaluation of both closed-source and open-source LLMs highlighted the effectiveness of prompting strategies, fine-tuning, and data augmentation in enhancing model performance in low-resource scenarios.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>LLMs</kwd>
        <kwd>large language models</kwd>
        <kwd>data augmentation</kwd>
        <kwd>EHR</kwd>
        <kwd>electronic health record</kwd>
        <kwd>comprehension</kwd>
        <kwd>patient education</kwd>
        <kwd>patient engagement</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <sec>
        <title>Background</title>
        <p>Electronic health record (EHR) notes serve as valuable sources of information that can significantly benefit patients. Programs like OpenNotes [<xref ref-type="bibr" rid="ref1">1</xref>] and the Blue Button [<xref ref-type="bibr" rid="ref2">2</xref>] initiative empower patients by providing access to their EHR notes [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref9">9</xref>]. Nevertheless, the benefits of accessing EHR notes can diminish if patients do not comprehend their content [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref15">15</xref>]. EHR notes are lengthy and filled with medical jargon [<xref ref-type="bibr" rid="ref16">16</xref>-<xref ref-type="bibr" rid="ref19">19</xref>], which can be difficult to comprehend for the average US adult, whose reading ability is around the seventh- to eighth-grade level [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref24">24</xref>]. Therefore, supportive technologies are needed to assist patients in understanding EHR content [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>], focusing on linking medical terms to lay-friendly terms [<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref29">29</xref>], consumer-oriented definitions [<xref ref-type="bibr" rid="ref18">18</xref>], and educational materials [<xref ref-type="bibr" rid="ref30">30</xref>]. Early studies have demonstrated that such interventions significantly enhance patient understanding [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]. However, initial methods primarily relied on frequency- and context-based approaches to identify unfamiliar terms and propose simpler synonyms [<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref29">29</xref>]. Identifying and extracting complex medical jargon from EHR notes is a crucial step toward improving patients’ comprehension, ultimately enhancing patient engagement and reducing anxiety about their health [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref31">31</xref>-<xref ref-type="bibr" rid="ref33">33</xref>].</p>
        <p>Notably, not all medical jargon extracted from EHR notes holds equal clinical importance [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]. Existing tools, such as MetaMap [<xref ref-type="bibr" rid="ref36">36</xref>], ScispaCy [<xref ref-type="bibr" rid="ref37">37</xref>], medspaCy [<xref ref-type="bibr" rid="ref38">38</xref>], and QuickUMLS [<xref ref-type="bibr" rid="ref39">39</xref>], are effective at extracting medical terms, typically predefined terms from the Unified Medical Language System (UMLS) [<xref ref-type="bibr" rid="ref40">40</xref>], but often fail to prioritize these terms based on their relevance to individual patients, treating all terms as equally important [<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref29">29</xref>]. In previous work, we asked physicians to identify medical jargon terms from EHR notes that are important to patients [<xref ref-type="bibr" rid="ref34">34</xref>]. Our results showed that physicians were able to consistently identify 5 to 10 medical jargon terms from each EHR note and rank each term based on its importance to patients [<xref ref-type="bibr" rid="ref34">34</xref>]. Furthermore, we developed feature-rich traditional machine learning models (eg, support vector machines) to identify such terms [<xref ref-type="bibr" rid="ref34">34</xref>]. However, our previous work did not focus on ranking jargon terms based on their importance to individual patients within an EHR note.</p>
        <p>In this study, we propose large language model (LLM)-based natural language processing approaches to identify and rank jargon terms from EHR notes based on their importance to patients. LLMs have demonstrated tremendous promise in biomedical natural language processing applications [<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref52">52</xref>] due to their exceptional generalizability and performance. However, applications in medical term extraction have primarily focused on tasks such as biomedical named entity recognition (BioNER) [<xref ref-type="bibr" rid="ref53">53</xref>-<xref ref-type="bibr" rid="ref55">55</xref>], rather than on prioritizing terms most relevant to patients, which is crucial for enhancing communication between patients and health care providers.</p>
        <p>The key contributions of this study are as follows:</p>
        <list list-type="order">
          <list-item>
            <p>We conduct a comprehensive evaluation of both closed-source and open-source LLMs to assess their effectiveness in identifying medical jargon from EHR notes that are important for patients using a physician-annotated gold-standard dataset.</p>
          </list-item>
          <list-item>
            <p>We leverage data augmentation with AI-generated medical jargon from Medical Information Mart for Intensive Care IV (MIMIC-IV) discharge summaries to address the challenges of training in low-resource settings. Under relaxed string matching, none of the methods surpassed the performance of closed-source models. In contrast, evaluation using the Jaccard Index showed that fine-tuning and data augmentation improved performance, exceeding that of closed-source models.</p>
          </list-item>
          <list-item>
            <p>We provide an in-depth analysis of the results from both quantitative and qualitative perspectives, focusing on common strategies for improving LLM performance, such as zero-shot and few-shot learning, prompt engineering, scaling laws, domain-adaptive training, and data augmentation, and recommendations for users based on our findings.</p>
          </list-item>
        </list>
      </sec>
      <sec>
        <title>Related Work</title>
        <p>Identifying jargon terms important to patients is part of the BioNER task, which involves identifying predefined entities in a text and labeling each token with the corresponding entity. Medical entities encompass categories such as diseases, medications, treatments, laboratory tests, and more [<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref57">57</xref>]. Studies such as [<xref ref-type="bibr" rid="ref58">58</xref>-<xref ref-type="bibr" rid="ref61">61</xref>] have introduced language models for BioNER tasks, while more recent studies [<xref ref-type="bibr" rid="ref53">53</xref>-<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref63">63</xref>] have explored the application of LLMs in BioNER. However, BioNER tasks primarily focus on extracting entities without considering their importance and relevance to the personal needs of patients, which distinguishes them from our objective.</p>
        <p>MedJex [<xref ref-type="bibr" rid="ref32">32</xref>] fine-tunes pretrained language models, such as Bidirectional Encoder Representations from Transformers (BERT [<xref ref-type="bibr" rid="ref64">64</xref>]), Robustly Optimized BERT Pretraining approach (RoBERTa [<xref ref-type="bibr" rid="ref59">59</xref>]), BioClinicalBERT [<xref ref-type="bibr" rid="ref65">65</xref>], and BioBERT [<xref ref-type="bibr" rid="ref58">58</xref>], on a domain-specific corpus. It leverages Wikipedia hyperlink spans during pretraining and transfers the learned weights to a target model fine-tuned on MedJ, an expert-annotated medical dataset. More recent studies [<xref ref-type="bibr" rid="ref66">66</xref>] have investigated whether LLMs, such as ChatGPT [<xref ref-type="bibr" rid="ref67">67</xref>], can outperform baseline pretrained language models (eg, MedJEx [<xref ref-type="bibr" rid="ref32">32</xref>] and SciSpacy [<xref ref-type="bibr" rid="ref37">37</xref>]) in extracting personalized medical jargon. Similarly, GAMedX [<xref ref-type="bibr" rid="ref68">68</xref>], a medical data extractor using LLMs (Mistral 7B and Gemma 7B), uses chained prompts to navigate the complexities of specialized medical jargon. Other works [<xref ref-type="bibr" rid="ref69">69</xref>-<xref ref-type="bibr" rid="ref71">71</xref>] have demonstrated how LLMs can enhance the readability of EHR notes by extracting medical jargon.</p>
        <p>This work also shares similarities with topic modeling, a task that extracts topics from input text. Using unsupervised learning algorithms, topic modeling can identify both explicit and implicit themes within a text corpus [<xref ref-type="bibr" rid="ref72">72</xref>-<xref ref-type="bibr" rid="ref74">74</xref>]. Through topic modeling, a text can be represented by multiple keywords or topics, which can then be incorporated into supervised models. However, topic modeling heavily relies on term frequencies and may easily overlook important terms that are clinically relevant to individual patients.</p>
        <p>Among the most relevant works, such as FOCUS [<xref ref-type="bibr" rid="ref34">34</xref>], ADS [<xref ref-type="bibr" rid="ref75">75</xref>], and FIT [<xref ref-type="bibr" rid="ref35">35</xref>], FOCUS [<xref ref-type="bibr" rid="ref34">34</xref>] uses MetaMap [<xref ref-type="bibr" rid="ref36">36</xref>] to extract medical jargon from EHR notes and uses feature-rich learning-to-rank techniques to determine whether the terms are important. However, none of the previous works have identified and ranked medical jargon terms in a note-specific manner. This is an important task, as ranking terms based on their relevance to a specific note may help the patient comprehend the note by linking important jargon terms to their lay definitions [<xref ref-type="bibr" rid="ref76">76</xref>] or help generate patient-friendly after-visit summaries [<xref ref-type="bibr" rid="ref77">77</xref>].</p>
      </sec>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Overview</title>
        <p>We evaluate the performance of the LLMs in three distinctive settings: (1) we assess the performance of closed- and open-source models by varying prompts and extraction tasks using the 10-fold annotated medical note (gold-standard dataset); (2) next, we benchmark the open-source models fine-tuned on portions of the gold-standard dataset; and (3) finally, we apply data augmentation, generating synthetic data from GPT-4o to fine-tune the open-source LLMs and evaluate them under the same varying settings. Our study finds that using data augmentation, the models can reach comparable or even superior performance in personalized medical jargon extraction tasks.</p>
        <p>We evaluated both closed-source and open-source LLMs for their efficacy in extracting key information from annotated medical notes, aiming to assess performance across different strategies. <xref rid="figure1" ref-type="fig">Figure 1</xref> provides an overview of our experiments, which leverage physician-annotated medical notes, closed- and open-source LLMs, and in-context learning (ICL). We examined the effects of prompting styles, fine-tuning, and data augmentation to enhance model performance.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>The evaluation workflow for closed- and open-source large language models (LLMs).</p>
          </caption>
          <graphic xlink:href="ai_v5i1e75561_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Data Source</title>
        <p>Our gold-standard dataset consists of 90 medical notes, each annotated by two physicians [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]. For each note, agreement from both physicians was used as the final annotation. The annotation agreement (micro average) on these notes was 0.51 Cohen Kappa [<xref ref-type="bibr" rid="ref35">35</xref>]. This EHR note dataset comprises text reports across six medical categories: cancer, chronic obstructive pulmonary disease, diabetes, heart failure, hypertension, and liver failure (<xref ref-type="table" rid="table1">Table 1</xref>). Each medical note includes detailed patient information and is accompanied by physician annotations highlighting the most critical terms or phrases relevant to the patient’s health status. <xref rid="figure2" ref-type="fig">Figure 2</xref> presents a snippet of a sample EHR note from the gold-standard dataset.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Gold-standard dataset description. The dataset consists of 90 notes from patients diagnosed with cancer, chronic obstructive pulmonary disease, diabetes, hypertension, liver failure, and heart failure.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="320"/>
            <col width="260"/>
            <col width="420"/>
            <thead>
              <tr valign="top">
                <td>Main diagnosis</td>
                <td>Note counts</td>
                <td>Median (IQR) jargon counts</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Cancer</td>
                <td>17</td>
                <td>8 (5-19)</td>
              </tr>
              <tr valign="top">
                <td>COPD<sup>a</sup></td>
                <td>20</td>
                <td>8 (5.8-13)</td>
              </tr>
              <tr valign="top">
                <td>Diabetes</td>
                <td>15</td>
                <td>8 (6-12.5)</td>
              </tr>
              <tr valign="top">
                <td>Hypertension</td>
                <td>11</td>
                <td>6 (4.5-8)</td>
              </tr>
              <tr valign="top">
                <td>Liver failure</td>
                <td>9</td>
                <td>5 (4-10)</td>
              </tr>
              <tr valign="top">
                <td>Heart failure</td>
                <td>18</td>
                <td>7 (5-10)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>COPD: chronic obstructive pulmonary disease.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>A sample electronic health record note where physicians identified important medical terms. Diagnoses or conditions are highlighted in yellow, while medications, tests, and procedures associated with those diagnoses are marked in green, accompanied by their respective rankings. EHR: electronic health record.</p>
          </caption>
          <graphic xlink:href="ai_v5i1e75561_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Closed-Source and Open-Source LLMs</title>
        <p>We used both publicly available LLMs (open-source LLMs) and proprietary models that are not publicly available (closed-source LLMs). The open-source LLMs included Mistral 7B v0.1 [<xref ref-type="bibr" rid="ref78">78</xref>] ( Mistral 7B), BioMistral 7B [<xref ref-type="bibr" rid="ref79">79</xref>], Llama3.1-8B [<xref ref-type="bibr" rid="ref80">80</xref>] (Llama 3.1 8B), DeepSeek-R1-Llama-Distill-8B [<xref ref-type="bibr" rid="ref81">81</xref>] (DeepSeek 8B). For proprietary models, we used 2 closed-source LLMs, which were from OpenAI [<xref ref-type="bibr" rid="ref67">67</xref>]: GPT-5.2 and GPT-5-mini.</p>
      </sec>
      <sec>
        <title>Zero-Shot Versus Few-Shot Prompts</title>
        <p>We used both zero-shot and few-shot prompts, also known as ICL, to compare the performance of the LLMs. For zero-shot prompting, we provided the model with general instructions, whereas for few-shot prompting, we included two examples randomly selected from the gold-standard dataset.</p>
      </sec>
      <sec>
        <title>General and Structured Prompts</title>
        <p>To evaluate the model’s performance, we implemented two variations of prompts: general prompts and structured prompts, each designed to assess how effectively the model could extract relevant clinical information from medical notes. <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> presents an example of a structured prompt. In this approach, the prompts are explicitly designed to closely align with the original task defined in the gold-standard dataset. The structured prompts instruct the LLMs to extract key medical conditions or diagnoses, followed by the relevant medications associated with them. In contrast, general prompts provide a more flexible and broader context. These prompts instruct the model to extract key medical terms without explicitly differentiating between conditions and medications, assigning the same base rank to both. This approach allows the LLM to interpret the extraction task more broadly, offering insights into how well the model generalizes its understanding of medical terminology when provided with less specific guidance.</p>
      </sec>
      <sec>
        <title>Fine-Tuning With Low-Rank Adaptation</title>
        <p>To improve the performance of open-source LLMs (Mistral 7B, BioMistral 7B, Llama 3.1 8B, and DeepSeek 8B), we conducted low-rank adaptation [<xref ref-type="bibr" rid="ref82">82</xref>] based on parameter-efficient fine-tuning. Low-rank adaptation is an efficient fine-tuning technique that allows models to be adapted to specific tasks without the need to update all of the model’s parameters. Instead, it applies low-rank updates to specific layers, reducing the computational cost and memory usage typically associated with traditional fine-tuning methods. The training was done with a batch size of 1 per device and gradient accumulation over 128 steps, and the low-rank dimension was set to 64. The learning rate was configured at 3<italic>e</italic> − 4, and the models were trained over 5 epochs to allow the models to converge effectively on the task-specific patterns present in the dataset.</p>
      </sec>
      <sec>
        <title>Data Augmentation</title>
        <p>Annotation by domain experts is expensive, and data augmentation using AI-generated data can help alleviate this challenge [<xref ref-type="bibr" rid="ref83">83</xref>]. Few-shot prompting integrates task examples directly into input prompts, allowing models to observe patterns and generalize from limited examples to effectively handle new, unseen data [<xref ref-type="bibr" rid="ref84">84</xref>]. We created an augmented dataset using the ICL technique, which was then used to refine the models (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). This augmented dataset was derived from discharge notes in the MIMIC-IV clinical database [<xref ref-type="bibr" rid="ref85">85</xref>]. A subset of MIMIC-IV discharge notes was randomly selected, and GPT-4o from OpenAI [<xref ref-type="bibr" rid="ref67">67</xref>] was used to process and rank key terms based on their importance for patient understanding. A dataset generated via GPT-4o was used because its outputs aligned more closely with the original annotations, exhibiting higher reliability during manual validation. The extraction process was guided by the examples, where two annotated notes from our gold-standard dataset were provided as examples to instruct the model on identifying and prioritizing terms.</p>
        <p>After the raw generation, we filtered out terms that do not appear in the medical text using string matching. The resulting augmented notes were 9995 notes. Using only a few-shot technique and no other filtering mechanism, we anticipated a lot of noise being injected into the augmented dataset. To this end, we further conducted an analysis on the augmented dataset, collecting 100 random cases and comparing the synthetic annotations with an expert annotation. We asked a clinician to annotate the MIMIC-IV notes with the same instructions given to the GPT-4o and compared the agreement between them.</p>
      </sec>
      <sec>
        <title>Exploring the Size of Augmented Dataset</title>
        <p>To explore the effects of augmented dataset size, we progressively increased the dataset across four scales: 10, 100, 1000, and 9995 notes. By testing the original gold-standard annotations along with AI-generated augmented data, we conducted a comprehensive evaluation of model performance across varying data sizes. Our objective was to enhance the robustness of LLMs in extracting critical medical information for patients.</p>
      </sec>
      <sec>
        <title>Baseline Models</title>
        <p>We evaluated the performance of the open-source LLMs against several prominent baselines in the field of named entity recognition tasks: MedJEx [<xref ref-type="bibr" rid="ref32">32</xref>] and BioClinical-ModernBERT [<xref ref-type="bibr" rid="ref86">86</xref>]. For MedJEx, we did not fine-tune the model since it is already a trained model to extract medical jargon. Also, MedJEx does not output any confidence score; therefore, we could only report precision, recall, and <italic>F</italic><sub>1</sub>-score. For BioClinical-ModernBERT, we fine-tuned the model on our dataset and measured the performance. BioClinical-ModernBERT also outputs confidence scores. We used it as a proxy for ranking, assigning higher rankings for higher confidence scores. For the candidate entity extraction, we used MedspaCy [<xref ref-type="bibr" rid="ref38">38</xref>] and QuickUMLS [<xref ref-type="bibr" rid="ref39">39</xref>] to form inputs for the baseline models.</p>
      </sec>
      <sec>
        <title>Evaluation Metrics</title>
        <p>We used precision (Equation 1), recall (Equation 2), <italic>F</italic><sub>1</sub>-score (Equation 3), and mean reciprocal rank (MRR; Equation 4) as our metrics. Performance metrics, including precision, recall, and <italic>F</italic><sub>1</sub>-score, were first calculated at the individual note level (). To aggregate these into a single representative score for each of the 10 folds, we calculated the macroaverage across all notes within that fold. The final results reported are the mean and SD of these fold-level averages. Precision and recall gave insights into the model’s ability to cover the annotated terms. The macro-<italic>F</italic><sub>1</sub>-score allowed us to evaluate the model’s balanced performance. MRR metric (Equation 4), in particular, evaluated how well the model ranked the terms according to their relevance, reflecting the model’s alignment with the annotated order of terms. Precision and recall are provided in the <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. For MRR calculation, integer-valued ranks were used for both outputs from using the general prompt and the structured prompt. Because the structured prompt outputs used integer values in the MRR calculation, some results could receive tied ranks. To avoid inflating MRR, we used average ranks for tied results (<italic>Adjusted Rank</italic><sub>i</sub>). Specifically, when multiple terms shared the same rank, each term was assigned the average of the positions they occupied. This provides a more conservative evaluation by penalizing ambiguous predictions and rewarding models that rank the correct term more specifically. We used two different string matching approaches: (1) relaxed string matching, which checks the matching as true positives when it either perfectly matches or contains a gold-standard input [<xref ref-type="bibr" rid="ref87">87</xref>,<xref ref-type="bibr" rid="ref88">88</xref>], and (2) the Jaccard Index method that uses set-based similarity to identify matches, using Jaccard Distance to measure the gap between two strings. Here, we set the similarity threshold to &#62;0.5. Any extracted term achieving a Jaccard similarity score greater than 0.5 against the gold-standard term was counted as a true positive for the subsequent precision and recall calculations. While the Jaccard Index method penalizes verbosity, relaxed string matching does not. This enables us to have a clear picture of the extraction performance and quality.</p>
        <graphic xlink:href="ai_v5i1e75561_fig6.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        <graphic xlink:href="ai_v5i1e75561_fig7.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        <graphic xlink:href="ai_v5i1e75561_fig8.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        <graphic xlink:href="ai_v5i1e75561_fig9.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        <graphic xlink:href="ai_v5i1e75561_fig10.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        <p>Here, <italic>RR</italic><sub>i</sub> denotes the reciprocal rank for each note <italic>i</italic>. <italic>k</italic> is the starting rank or the tied rank, and <italic>v</italic> is the count of ties (Equation 5). As shown in Equation 5, we compute the arithmetic mean of the ordinal positions occupied by tied terms to determine the effective rank. The final MRR is then derived by averaging these adjusted <italic>RR</italic><sub>i</sub> values across the total number of notes <italic>N</italic> in the dataset (Equation 4).</p>
      </sec>
      <sec>
        <title>Error Classification</title>
        <p>To further understand the nature of the model’s behavior, we conducted an error analysis of the model’s outputs. First, we classified the model’s outputs that were erroneous. We used three types of errors: (1) spurious error (SE), where the model included terms that are not in the gold-standard dataset but are included in the medical note; (2) granularity error (GE), which reflects overly specific or overly broad extractions; and (3) misalignment error (ME), where the model simply did not understand the instruction and misbehaved. Using this taxonomy, we identified the erroneous behavior of the LLMs.</p>
      </sec>
      <sec>
        <title>Experimental Details</title>
        <p>All experiments were performed with two Nvidia A100 graphics processing units, each with 40 GB of memory, an Intel Xeon Gold 6230 CPU, and 192 GB of RAM. We used Python 3.9 and the Hugging Face transformers library [<xref ref-type="bibr" rid="ref89">89</xref>] for our experiment. For the closed-source models, we used OpenAI’s ChatGPT API [<xref ref-type="bibr" rid="ref67">67</xref>]. To reduce the randomness of the experiment, we set the generation temperature to 0.1 for every model.</p>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>The requirement for ethical approval and informed consent was waived by the institutional review board at the VA Bedford Health Care System. The experiments were performed in accordance with the Declaration of Helsinki. The clinical data used for data augmentation were obtained from the MIMIC-IV database, a publicly available, deidentified repository of EHRs from patients admitted to the emergency department or an intensive care unit at the Beth Israel Deaconess Medical Center in Boston, Massachusetts. Access to the database was granted following the completion of the required PhysioNet Credentialed Health Data Use Agreement 1.5.0.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <p>The highest baseline <italic>F</italic><sub>1</sub>-score was achieved by MedJEx (0.138, SD 0.031) using the Jaccard Index. For MRR, BioClinical-ModernBERT scored 0.087 (SD 0.146) and 0.098 (SD 0.151) using relaxed string matching and Jaccard Index, respectively. Despite being fine-tuned on the dataset, BioClinical-ModernBERT did not show the highest performance. As we expected, the conventional methods have limitations in extracting medical entities that are particularly important to the patient. Given that these models have a parameter size of 150M, considerably smaller than the generative models used for comparison, this outcome is perhaps unsurprising. <xref ref-type="table" rid="table2">Table 2</xref> presents the performance of the baseline models, which include commonly used language models for medical information extraction.</p>
      <table-wrap position="float" id="table2">
        <label>Table 2</label>
        <caption>
          <p>Performance of baseline models of F1-score and mean reciprocal rank using relaxed string matching (relaxed) and Jaccard Distance–based method (Jaccard).</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="230"/>
          <col width="210"/>
          <col width="210"/>
          <col width="180"/>
          <col width="170"/>
          <thead>
            <tr valign="bottom">
              <td>Models</td>
              <td><italic>F</italic><sub>1</sub>-score (relaxed), mean (SD)</td>
              <td><italic>F</italic><sub>1</sub>-score (Jaccard), mean (SD)</td>
              <td>MRR<sup>a</sup> (SD), relaxed</td>
              <td>MRR (SD), Jaccard</td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td>MedJEx</td>
              <td>0.120 (0.027)</td>
              <td>0.138 (0.031)</td>
              <td>—<sup>b</sup></td>
              <td>—</td>
            </tr>
            <tr valign="top">
              <td>BioClinical-Modern BERT</td>
              <td>0.076 (0.079)</td>
              <td>0.087 (0.086)</td>
              <td>0.087 (0.146)</td>
              <td>0.098 (0.151)</td>
            </tr>
          </tbody>
        </table>
        <table-wrap-foot>
          <fn id="table2fn1">
            <p><sup>a</sup>MRR: mean reciprocal rank.</p>
          </fn>
          <fn id="table2fn2">
            <p><sup>b</sup>Not available.</p>
          </fn>
        </table-wrap-foot>
      </table-wrap>
      <p>The highest <italic>F</italic><sub>1</sub>-score in the zero-shot setting was 0.496 (SD 0.058), achieved by GPT-5.2. GPT-5.2 achieved the highest MRR in zero-shot prompts (0.578, SD 0.045) and (0.467, SD 0.117) with relaxed string matching and Jaccard Index. Notably, DeepSeek 8B and Llama 3.1 8B achieved higher Jaccard Index scores than GPT-5.2. As this metric is inversely affected by verbosity—additional non-overlapping tokens increase the union while leaving the intersection unchanged—GPT-5.2’s more expansive outputs likely explain its lower performance. We also observed a substantial gap between relaxed string matching and Jaccard Index scores across models. For example, BioMistral 7B (<xref ref-type="table" rid="table3">Tables 3</xref> and <xref ref-type="table" rid="table4">4</xref>) exhibited a notable discrepancy in <italic>F</italic><sub>1</sub>-scores between these 2 metrics, indicating that vanilla models tended to produce verbose outputs. <xref ref-type="table" rid="table3">Tables 3</xref> and <xref ref-type="table" rid="table4">4</xref> present the results from both closed- and open-source models using zero-shot and few-shot prompts, respectively. In <xref ref-type="table" rid="table3">Table 3</xref>, we compared the performance of different models using the top 5 and top 10 results for <italic>F</italic><sub>1</sub>-score and MRR across both closed- and open-source LLMs, providing their mean and SD. GPT-5.2 showed the highest performance of <italic>F</italic><sub>1</sub>-score and MRR using relaxed string matching, but DeepSeek 8B also showed the highest <italic>F</italic><sub>1</sub>-score using the Jaccard Index. In <xref ref-type="table" rid="table4">Table 4</xref>, we compared the performance of different models using the top 5 and top 10 results for <italic>F</italic><sub>1</sub>-score and MRR across both closed- and open-source LLMs, providing their mean and SD. GPT-5.2 achieved the best <italic>F</italic><sub>1</sub>-score and MRR score using relaxed string matching. However, Mistral 7B also achieved a better <italic>F</italic><sub>1</sub>-score than GPT-5.2 using the Jaccard Index.</p>
      <table-wrap position="float" id="table3">
        <label>Table 3</label>
        <caption>
          <p>Comparison between closed- and open-source vanilla models with zero-shot prompts.</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="30"/>
          <col width="30"/>
          <col width="0"/>
          <col width="110"/>
          <col width="0"/>
          <col width="100"/>
          <col width="0"/>
          <col width="100"/>
          <col width="0"/>
          <col width="90"/>
          <col width="0"/>
          <col width="130"/>
          <col width="0"/>
          <col width="0"/>
          <col width="120"/>
          <col width="0"/>
          <col width="100"/>
          <col width="0"/>
          <col width="90"/>
          <col width="0"/>
          <col width="100"/>
          <thead>
            <tr valign="top">
              <td colspan="3">Models</td>
              <td colspan="2">Prompt</td>
              <td colspan="9">Top 5</td>
              <td colspan="7">Top 10</td>
            </tr>
            <tr valign="bottom">
              <td colspan="3">
                <break/>
              </td>
              <td colspan="2">
                <break/>
              </td>
              <td colspan="2"><italic>F</italic><sub>1</sub>-score (relaxed), mean (SD)</td>
              <td colspan="2"><italic>F</italic><sub>1</sub>-score (Jaccard), mean (SD)</td>
              <td colspan="2">MRR<sup>a</sup> (SD), relaxed</td>
              <td colspan="2">MRR (SD), Jaccard</td>
              <td colspan="3"><italic>F</italic><sub>1</sub>-score (relaxed), mean (SD)</td>
              <td colspan="2"><italic>F</italic><sub>1</sub>-score (Jaccard), mean (SD)</td>
              <td colspan="2">MRR, SD (relaxed)</td>
              <td>MRR, SD (Jaccard)</td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td colspan="21">
                <bold>Closed-source LLMs</bold>
                <sup>b</sup>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="20">
                <bold>GPT-5.2</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">General</td>
              <td colspan="2">0.492 (0.068)<sup>c</sup></td>
              <td colspan="2">0.307 (0.071)</td>
              <td colspan="2">0.578 (0.045)<sup>c</sup></td>
              <td colspan="2">0.476 (0.124)<sup>c</sup></td>
              <td colspan="3">0.496 (0.058)<sup>c</sup></td>
              <td colspan="2">0.31 (0.077)</td>
              <td colspan="2">0.578 (0.045)<sup>c</sup></td>
              <td colspan="2">0.467 (0.117)<sup>c</sup></td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">Structured</td>
              <td colspan="2">0.332 (0.077)</td>
              <td colspan="2">0.206 (0.048)</td>
              <td colspan="2">0.513 (0.153)</td>
              <td colspan="2">0.359 (0.123)</td>
              <td colspan="3">0.336 (0.073)</td>
              <td colspan="2">0.211 (0.047)</td>
              <td colspan="2">0.513 (0.153)</td>
              <td colspan="2">0.359 (0.123)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="20">
                <bold>GPT-5-mini</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">General</td>
              <td colspan="2">0.46 (0.073)</td>
              <td colspan="2">0.129 (0.048)</td>
              <td colspan="2">0.561 (0.156)</td>
              <td colspan="2">0.196 (0.11)</td>
              <td colspan="3">0.505 (0.063)</td>
              <td colspan="2">0.145 (0.045)</td>
              <td colspan="2">0.552 (0.151)</td>
              <td colspan="2">0.196 (0.11)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">Structured</td>
              <td colspan="2">0.387 (0.055)</td>
              <td colspan="2">0.12 (0.046)</td>
              <td colspan="2">0.575 (0.165)</td>
              <td colspan="2">0.24 (0.17)</td>
              <td colspan="3">0.387 (0.075)</td>
              <td colspan="2">0.127 (0.052)</td>
              <td colspan="2">0.566 (0.153)</td>
              <td colspan="2">0.24 (0.17)</td>
            </tr>
            <tr valign="top">
              <td colspan="21">
                <bold>Open-source LLMs</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="20">
                <bold>Mistral 7B</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">General</td>
              <td colspan="2">0.372 (0.085)</td>
              <td colspan="2">0.323 (0.081)</td>
              <td colspan="2">0.451 (0.175)</td>
              <td colspan="2">0.453 (0.172)</td>
              <td colspan="3">0.401 (0.119)</td>
              <td colspan="2">0.345 (0.115)</td>
              <td colspan="2">0.432 (0.167)</td>
              <td colspan="2">0.444 (0.177)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">Structured</td>
              <td colspan="2">0.351 (0.083)</td>
              <td colspan="2">0.243 (0.056)</td>
              <td colspan="2">0.582 (0.108)</td>
              <td colspan="2">0.455 (0.112)</td>
              <td colspan="3">0.349 (0.081)</td>
              <td colspan="2">0.236 (0.055)</td>
              <td colspan="2">0.573 (0.106)</td>
              <td colspan="2">0.455 (0.112)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="20">
                <bold>Llama 3.1 8B</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">General</td>
              <td colspan="2">0.347 (0.054)</td>
              <td colspan="2">0.333 (0.055)</td>
              <td colspan="2">0.426 (0.112)</td>
              <td colspan="2">0.392 (0.108)</td>
              <td colspan="3">0.368 (0.063)</td>
              <td colspan="2">0.35 (0.056)</td>
              <td colspan="2">0.422 (0.113)</td>
              <td colspan="2">0.392 (0.108)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">Structured</td>
              <td colspan="2">0.371 (0.056)</td>
              <td colspan="2">0.188 (0.051)</td>
              <td colspan="2">0.521 (0.14)</td>
              <td colspan="2">0.318 (0.112)</td>
              <td colspan="3">0.358 (0.059)</td>
              <td colspan="2">0.186 (0.055)</td>
              <td colspan="2">0.517 (0.139)</td>
              <td colspan="2">0.318 (0.112)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="20">
                <bold>BioMistral 7B</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">General</td>
              <td colspan="2">0.21 (0.076)</td>
              <td colspan="2">0.017 (0.032)</td>
              <td colspan="2">0.353 (0.141)</td>
              <td colspan="2">0.033 (0.071)</td>
              <td colspan="3">0.211 (0.078)</td>
              <td colspan="2">0.02 (0.04)</td>
              <td colspan="2">0.353 (0.141)</td>
              <td colspan="2">0.033 (0.071)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">Structured</td>
              <td colspan="2">0.304 (0.121)</td>
              <td colspan="2">0.017 (0.018)</td>
              <td colspan="2">0.507 (0.152)</td>
              <td colspan="2">0.056 (0.056)</td>
              <td colspan="3">0.304 (0.123)</td>
              <td colspan="2">0.017 (0.018)</td>
              <td colspan="2">0.507 (0.152)</td>
              <td colspan="2">0.056 (0.056)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="20">
                <bold>DeepSeek 8B</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">General</td>
              <td colspan="2">0.42 (0.077)</td>
              <td colspan="2">0.383 (0.053)<sup>c</sup></td>
              <td colspan="2">0.416 (0.113)</td>
              <td colspan="2">0.398 (0.13)</td>
              <td colspan="3">0.443 (0.077)</td>
              <td colspan="2">0.409 (0.068)<sup>c</sup></td>
              <td colspan="2">0.414 (0.113)</td>
              <td colspan="2">0.397 (0.13)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td colspan="2">Structured</td>
              <td colspan="2">0.326 (0.026)</td>
              <td colspan="2">0.211 (0.053)</td>
              <td colspan="2">0.467 (0.131)</td>
              <td colspan="2">0.4 (0.158)</td>
              <td colspan="3">0.328 (0.048)</td>
              <td colspan="2">0.218 (0.059)</td>
              <td colspan="2">0.467 (0.131)</td>
              <td colspan="2">0.4 (0.158)</td>
            </tr>
          </tbody>
        </table>
        <table-wrap-foot>
          <fn id="table3fn1">
            <p><sup>a</sup>MRR: mean reciprocal rank.</p>
          </fn>
          <fn id="table3fn2">
            <p><sup>b</sup>Highest score for each metric (column) among the compared models.</p>
          </fn>
          <fn id="table3fn3">
            <p><sup>c</sup>LLM: large language model.</p>
          </fn>
        </table-wrap-foot>
      </table-wrap>
      <table-wrap position="float" id="table4">
        <label>Table 4</label>
        <caption>
          <p>Comparison between closed- and open-source vanilla models with few-shot prompts.</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="30"/>
          <col width="30"/>
          <col width="190"/>
          <col width="0"/>
          <col width="90"/>
          <col width="0"/>
          <col width="90"/>
          <col width="0"/>
          <col width="110"/>
          <col width="0"/>
          <col width="90"/>
          <col width="0"/>
          <col width="0"/>
          <col width="90"/>
          <col width="0"/>
          <col width="90"/>
          <col width="0"/>
          <col width="90"/>
          <col width="0"/>
          <col width="100"/>
          <thead>
            <tr valign="top">
              <td colspan="4">Models and prompts</td>
              <td colspan="9">Top 5</td>
              <td colspan="7">Top 10</td>
            </tr>
            <tr valign="bottom">
              <td colspan="4">
                <break/>
              </td>
              <td colspan="2"><italic>F</italic><sub>1</sub>-score, mean (SD), relaxed</td>
              <td colspan="2"><italic>F</italic><sub>1</sub>-score, mean (SD), Jaccard</td>
              <td colspan="2">MRR<sup>a</sup> (SD), relaxed</td>
              <td colspan="2">MRR (SD), Jaccard</td>
              <td colspan="3"><italic>F</italic><sub>1</sub>-score, mean (SD), relaxed</td>
              <td colspan="2"><italic>F</italic><sub>1</sub>-score, mean (SD), Jaccard</td>
              <td colspan="2">MRR (SD), relaxed</td>
              <td>MRR (SD), Jaccard</td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td colspan="20">
                <bold>Closed-source LLMs<sup>b</sup></bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="19">
                <bold>GPT-5.2</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>General</td>
              <td colspan="2">0.475 (0.068)<sup>c</sup></td>
              <td colspan="2">0.323 (0.06)</td>
              <td colspan="2">0.561 (0.098)<sup>c</sup></td>
              <td colspan="2">0.536 (0.092)<sup>c</sup></td>
              <td colspan="3">0.478 (0.07)<sup>c</sup></td>
              <td colspan="2">0.326 (0.058)</td>
              <td colspan="2">0.561 (0.098)<sup>c</sup></td>
              <td colspan="2">0.536 (0.092)<sup>c</sup></td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>Structured</td>
              <td colspan="2">0.354 (0.069)</td>
              <td colspan="2">0.233 (0.054)</td>
              <td colspan="2">0.536 (0.174)</td>
              <td colspan="2">0.387 (0.166)</td>
              <td colspan="3">0.364 (0.076)</td>
              <td colspan="2">0.237 (0.052)</td>
              <td colspan="2">0.536 (0.173)</td>
              <td colspan="2">0.387 (0.165)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="19">
                <bold>GPT-5-mini</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>General</td>
              <td colspan="2">0.475 (0.077)</td>
              <td colspan="2">0.243 (0.087)</td>
              <td colspan="2">0.543 (0.138)</td>
              <td colspan="2">0.4 (0.133)</td>
              <td colspan="3">0.492 (0.056)</td>
              <td colspan="2">0.245 (0.069)</td>
              <td colspan="2">0.543 (0.138)</td>
              <td colspan="2">0.4 (0.133)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>Structured</td>
              <td colspan="2">0.348 (0.068)</td>
              <td colspan="2">0.161 (0.041)</td>
              <td colspan="2">0.508 (0.14)</td>
              <td colspan="2">0.321 (0.172)</td>
              <td colspan="3">0.347 (0.087)</td>
              <td colspan="2">0.163 (0.052)</td>
              <td colspan="2">0.495 (0.131)</td>
              <td colspan="2">0.323 (0.174)</td>
            </tr>
            <tr valign="top">
              <td colspan="20">
                <bold>Open-source LLMs</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="19">
                <bold>Mistral 7B</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>General</td>
              <td colspan="2">0.387 (0.046)</td>
              <td colspan="2">0.338 (0.039)</td>
              <td colspan="2">0.498 (0.141)</td>
              <td colspan="2">0.503 (0.11)</td>
              <td colspan="3">0.383 (0.05)</td>
              <td colspan="2">0.335 (0.045)</td>
              <td colspan="2">0.498 (0.141)</td>
              <td colspan="2">0.503 (0.11)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>Structured</td>
              <td colspan="2">0.363 (0.067)</td>
              <td colspan="2">0.335 (0.056)</td>
              <td colspan="2">0.534 (0.104)</td>
              <td colspan="2">0.533 (0.101)</td>
              <td colspan="3">0.356 (0.073)</td>
              <td colspan="2">0.332 (0.065)</td>
              <td colspan="2">0.534 (0.104)</td>
              <td colspan="2">0.533 (0.101)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="19">
                <bold>Llama 3.1 8B</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>General</td>
              <td colspan="2">0.372 (0.078)</td>
              <td colspan="2">0.335 (0.093)</td>
              <td colspan="2">0.521 (0.145)</td>
              <td colspan="2">0.557 (0.126)</td>
              <td colspan="3">0.369 (0.073)</td>
              <td colspan="2">0.333 (0.096)</td>
              <td colspan="2">0.521 (0.145)</td>
              <td colspan="2">0.547 (0.132)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>Structured</td>
              <td colspan="2">0.368 (0.116)</td>
              <td colspan="2">0.346 (0.111)<sup>c</sup></td>
              <td colspan="2">0.442 (0.066)</td>
              <td colspan="2">0.502 (0.109)</td>
              <td colspan="3">0.374 (0.126)</td>
              <td colspan="2">0.353 (0.119)<sup>c</sup></td>
              <td colspan="2">0.442 (0.066)</td>
              <td colspan="2">0.502 (0.109)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="19">
                <bold>BioMistral 7B</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>General</td>
              <td colspan="2">0.208 (0.063)</td>
              <td colspan="2">0.08 (0.036)</td>
              <td colspan="2">0.459 (0.115)</td>
              <td colspan="2">0.3 (0.132)</td>
              <td colspan="3">0.207 (0.067)</td>
              <td colspan="2">0.081 (0.033)</td>
              <td colspan="2">0.459 (0.115)</td>
              <td colspan="2">0.3 (0.132)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>Structured</td>
              <td colspan="2">0.227 (0.085)</td>
              <td colspan="2">0.098 (0.042)</td>
              <td colspan="2">0.435 (0.072)</td>
              <td colspan="2">0.318 (0.136)</td>
              <td colspan="3">0.225 (0.084)</td>
              <td colspan="2">0.097 (0.041)</td>
              <td colspan="2">0.435 (0.072)</td>
              <td colspan="2">0.318 (0.136)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td colspan="19">
                <bold>DeepSeek 8B</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>General</td>
              <td colspan="2">0.344 (0.058)</td>
              <td colspan="2">0.335 (0.058)</td>
              <td colspan="2">0.494 (0.125)</td>
              <td colspan="2">0.485 (0.115)</td>
              <td colspan="3">0.344 (0.066)</td>
              <td colspan="2">0.334 (0.064)</td>
              <td colspan="2">0.485 (0.116)</td>
              <td colspan="2">0.476 (0.102)</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>
                <break/>
              </td>
              <td>Structured</td>
              <td colspan="2">0.327 (0.105)</td>
              <td colspan="2">0.299 (0.116)</td>
              <td colspan="2">0.48 (0.173)</td>
              <td colspan="2">0.466 (0.188)</td>
              <td colspan="3">0.332 (0.105)</td>
              <td colspan="2">0.304 (0.113)</td>
              <td colspan="2">0.48 (0.173)</td>
              <td colspan="2">0.466 (0.188)</td>
            </tr>
          </tbody>
        </table>
        <table-wrap-foot>
          <fn id="table4fn1">
            <p><sup>a</sup>MRR: mean reciprocal rank.</p>
          </fn>
          <fn id="table4fn2">
            <p><sup>b</sup>Highest score for each metric (column) among the compared models.</p>
          </fn>
          <fn id="table4fn3">
            <p><sup>c</sup>LLM: large language model.</p>
          </fn>
        </table-wrap-foot>
      </table-wrap>
      <p>There was no clear advantage of using a few-shot prompt. The highest scores were achieved by using zero-shot prompts. Although variations still exist, most of the models (GPT-5.2, GPT-5-mini, Mistral 7B, and DeepSeek 8B) showed higher scores using a general prompt over a structured prompt. We hypothesize that each LLM has its own preferred prompts.</p>
      <p>Fine-tuning open-source models resulted in better performance compared to vanilla models. Here, we unified the prompt style into a general prompt to evaluate the effectiveness of fine-tuning. The best-performing model was DeepSeek 8B, showing 0.424 (SD 0.052) in <italic>F</italic><sub>1</sub>-score (relaxed) and 0.431 (SD 0.046) in <italic>F</italic><sub>1</sub>-score (Jaccard) in the top 10 medical jargon extraction. For MRR, BioMistral 7B showed the best performance of 0.565 (SD 0.111) in MRR (relaxed) and 0.577 (SD 0.109) in MRR (Jaccard). Under relaxed string matching, fine-tuning did not outperform closed-source models such as GPT-5 (<xref ref-type="table" rid="table3">Tables 3</xref>-<xref ref-type="table" rid="table5">5</xref>). However, it yielded notable improvements in domain-specific task performance. Conversely, fine-tuned models exceeded closed-source models under the Jaccard Index metric, indicating that fine-tuning reduced verbosity while maintaining relevant content in the generated outputs. <xref ref-type="table" rid="table5">Table 5</xref> demonstrates a reduced margin between the relaxed and Jaccard metrics. In <xref ref-type="table" rid="table5">Table 5</xref>, we used zero-shot prompting with a general prompting style in the top 5 and top 10 medical jargon extraction. In the top 10 extraction task, DeepSeek 8B presented the highest <italic>F</italic><sub>1</sub>-score in both metrics (relaxed and Jaccard). However, BioMistral 7B showed the highest MRR score in both metrics.</p>
      <table-wrap position="float" id="table5">
        <label>Table 5</label>
        <caption>
          <p>The average F1-score and mean reciprocal rank score of fine-tuned open-source models with 10-fold cross-validation.</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="140"/>
          <col width="110"/>
          <col width="110"/>
          <col width="110"/>
          <col width="110"/>
          <col width="0"/>
          <col width="100"/>
          <col width="110"/>
          <col width="110"/>
          <col width="100"/>
          <thead>
            <tr valign="top">
              <td>Models</td>
              <td colspan="5">Top 5</td>
              <td colspan="4">Top 10</td>
            </tr>
            <tr valign="bottom">
              <td>
                <break/>
              </td>
              <td><italic>F</italic><sub>1</sub>-score, mean (SD), relaxed</td>
              <td><italic>F</italic><sub>1</sub>-score, mean (SD), Jaccard</td>
              <td>MRR<sup>a</sup> (SD), relaxed</td>
              <td>MRR (SD), Jaccard</td>
              <td colspan="2"><italic>F</italic><sub>1</sub>-score, mean (SD), relaxed</td>
              <td><italic>F</italic><sub>1</sub>-score, mean (SD), Jaccard</td>
              <td>MRR (SD), relaxed</td>
              <td>MRR (SD), Jaccard</td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td>Mistral 7B</td>
              <td>0.417 (0.093)</td>
              <td>0.43 (0.074)<sup>b</sup></td>
              <td>0.525 (0.11)</td>
              <td>0.573 (0.111)</td>
              <td colspan="2">0.416 (0.093)</td>
              <td>0.428 (0.074)</td>
              <td>0.525 (0.11)</td>
              <td>0.573 (0.111)</td>
            </tr>
            <tr valign="top">
              <td>Llama 3.1 8B</td>
              <td>0.38 (0.065)</td>
              <td>0.384 (0.058)</td>
              <td>0.515 (0.078)</td>
              <td>0.527 (0.103)</td>
              <td colspan="2">0.384 (0.07)</td>
              <td>0.388 (0.064)</td>
              <td>0.515 (0.078)</td>
              <td>0.527 (0.103)</td>
            </tr>
            <tr valign="top">
              <td>BioMistral 7B</td>
              <td>0.374 (0.07)</td>
              <td>0.379 (0.062)</td>
              <td>0.565 (0.111)<sup>b</sup></td>
              <td>0.577 (0.109)<sup>b</sup></td>
              <td colspan="2">0.377 (0.07)</td>
              <td>0.381 (0.061)</td>
              <td>0.565 (0.111)<sup>b</sup></td>
              <td>0.577 (0.109)<sup>b</sup></td>
            </tr>
            <tr valign="top">
              <td>DeepSeek 8B</td>
              <td>0.42 (0.048)<sup>b</sup></td>
              <td>0.429 (0.043)</td>
              <td>0.508 (0.085)</td>
              <td>0.501 (0.075)</td>
              <td colspan="2">0.424 (0.052)<sup>b</sup></td>
              <td>0.431 (0.046)<sup>b</sup></td>
              <td>0.508 (0.085)</td>
              <td>0.501 (0.075)</td>
            </tr>
          </tbody>
        </table>
        <table-wrap-foot>
          <fn id="table5fn1">
            <p><sup>a</sup>MRR: mean reciprocal rank.</p>
          </fn>
          <fn id="table5fn2">
            <p><sup>b</sup>Highest score for each metric (column) among the compared models.</p>
          </fn>
        </table-wrap-foot>
      </table-wrap>
      <p>The highest-performing model that used a data augmentation strategy was DeepSeek 8B, achieving an <italic>F</italic><sub>1</sub>-score (relaxed) above 0.425 and an <italic>F</italic><sub>1</sub>-score (Jaccard) of 0.401 in the top 10 medical jargon extractions (<xref rid="figure3" ref-type="fig">Figure 3</xref>; Tables S1 and S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). The highest MRR score was achieved by DeepSeek 8B in the top 5 and top 10 medical jargon extractions using both metrics (relaxed and Jaccard). Performance patterns varied depending on the models and the metric. For example, under relaxed string matching, BioMistral 7B showed an increasing <italic>F</italic><sub>1</sub>-score trend, but MRR peaked at n=1000 before declining in both top 5 and top 10 extraction tasks. In contrast, Mistral 7B achieved peak performance at n=100 for top 5 extraction, while DeepSeek performed best at n=10 and n=9995. However, when evaluated using the Jaccard Index, all models exhibited consistently increasing <italic>F</italic><sub>1</sub>-scores as dataset size increased. These findings are discussed further below. Although the results failed to achieve higher scores than closed-source models under relaxed string matching, using the augmented dataset was found useful in some models, such as BioMistral 7B, Llama 3.1 8B, and DeepSeek 8B, but not in Mistral 7B. However, under the Jaccard Index, our results show that data augmentation could also outperform closed-source models.</p>
      <fig id="figure3" position="float">
        <label>Figure 3</label>
        <caption>
          <p>Performance comparison of open-source large language models across varying scales of Medical Information Mart for Intensive Care IV–augmented data. Evaluation is based on relaxed string metrics (A) and Jaccard Distance (B), with training sizes ranging from 10 to 9995 samples. While individual model performance varies, a general trend of improvement is observed using the Jaccard Index (B) as the dataset size increases. Bar graphs represent F1-scores, while line graphs depict mean reciprocal rank. Both metrics indicate that DeepSeek 8B achieves the highest F1-score when trained on the largest dataset size and the highest mean reciprocal rank.</p>
        </caption>
        <graphic xlink:href="ai_v5i1e75561_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
      </fig>
      <p>To further investigate error distributions, we evaluated the performance of the vanilla DeepSeek model against its fine-tuned and augmented counterparts (n=9995) using a representative test set from our 10-fold cross-validation. As illustrated in <xref ref-type="table" rid="table6">Table 6</xref>, the vanilla DeepSeek model demonstrated the highest frequency of MEs and GEs. While fine-tuning substantially reduced all error types, most notably SEs, which dropped to 46, the total error count was lowest for the fine-tuned model. Conversely, the data augmentation strategy did not yield similar improvements; it resulted in 80 SEs, surpassing the vanilla model, and only marginal reductions in GEs and MEs. Ultimately, the data augmentation approach failed to achieve the high-magnitude error reduction observed with standard fine-tuning.</p>
      <table-wrap position="float" id="table6">
        <label>Table 6</label>
        <caption>
          <p>The error analysis table. All models show a high level of spurious error, but fine-tuning substantially reduces the error rate, showing the lowest total error counts among other methods.</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="220"/>
          <col width="190"/>
          <col width="200"/>
          <col width="200"/>
          <col width="190"/>
          <thead>
            <tr valign="top">
              <td>Model versions</td>
              <td>Spurious error, n</td>
              <td>Granularity error, n</td>
              <td>Misalignment error, n</td>
              <td>Total error count, n</td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td>Vanilla model</td>
              <td>76</td>
              <td>7</td>
              <td>37</td>
              <td>120</td>
            </tr>
            <tr valign="top">
              <td>Fine-tuned</td>
              <td>46</td>
              <td>1</td>
              <td>33</td>
              <td>80</td>
            </tr>
            <tr valign="top">
              <td>Augmented dataset</td>
              <td>80</td>
              <td>2</td>
              <td>33</td>
              <td>115</td>
            </tr>
          </tbody>
        </table>
      </table-wrap>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Zero-Shot vs Few-Shot</title>
        <p>Our findings suggest that there are minimal differences between zero-shot and few-shot prompting. Although there are some cases where a few-shot prompt outperformed a zero-shot prompt, these results are not always consistent. Results differ by model, and it is hard to tell which prompting, zero-shot or few-shot, performs better. This is consistent with the findings reported by Chamieh et al [<xref ref-type="bibr" rid="ref90">90</xref>], which found that few-shot prompting is not always more advantageous than zero-shot prompting (<xref rid="figure4" ref-type="fig">Figure 4</xref>).</p>
        <fig id="figure4" position="float">
          <label>Figure 4</label>
          <caption>
            <p>Case study for extracting the top 10 important medical jargons from Mistral 7B in zero-shot and few-shot settings. Interestingly, few-shot prompting fails to extract terms that clinicians annotated as “gold,” whereas zero-shot successfully conducts the task. COPD: chronic obstructive pulmonary disease; GERD: gastroesophageal reflux disease.</p>
          </caption>
          <graphic xlink:href="ai_v5i1e75561_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Prompting Styles</title>
        <p>The effectiveness of prompts can vary across models, with certain prompt styles enhancing performance for specific models (<xref ref-type="table" rid="table3">Table 3</xref>). While differences between models were generally minimal, some models performed better with particular prompt styles. For example, in the Llama 3.1 8B model, structured prompts outperformed general prompts with few-shot prompting, whereas in Mistral 7B and in other models, general prompts showed improved performance over structured prompts regardless of string matching metrics. For both Llama 3.1 8B and Mistral 7B, structured prompts induced higher recall than precision. Since a structured format gives more specific instructions, this may cause the model to generate concise and precise results (Tables S1 and S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). A plausible explanation is that structured prompts impose stronger output constraints, narrowing the model’s generation space. Largely, our results align with prior research suggesting that tailored prompts can optimize a model’s performance in specialized tasks [<xref ref-type="bibr" rid="ref91">91</xref>]. Interestingly, in most cases, we found that <italic>F</italic><sub>1</sub>-score trends are aligned with MRR scores, suggesting that certain prompts helped the model reach its full potential. This shows that testing diverse prompt styles is essential in maximizing model performance, which is highly aligned with the current research results [<xref ref-type="bibr" rid="ref92">92</xref>-<xref ref-type="bibr" rid="ref95">95</xref>].</p>
      </sec>
      <sec>
        <title>Fine-Tuning With Gold-Standard Data</title>
        <p>Fine-tuning LLMs using domain-specific data proved effective in enhancing model performance, even surpassing the performance of closed-source models (GPT-5) under the Jaccard Index (<xref ref-type="table" rid="table5">Table 5</xref>). The size of the datasets used for fine-tuning was small but led to performance gains. Increasing the size of the fine-tuning dataset is likely to further improve the model’s performance [<xref ref-type="bibr" rid="ref96">96</xref>]. <xref rid="figure5" ref-type="fig">Figure 5</xref> illustrates the impact of fine-tuning: while vanilla models show limitations in extracting key points using prompts alone, fine-tuning enables the models to learn relevant patterns, leading to improved performance. Although BioMistral 7B could extract some information with instructions, MEs were present in its output. After conducting fine-tuning, the model dramatically improved performance, reducing its errors as shown in <xref rid="figure5" ref-type="fig">Figure 5</xref>. This trend is further evidenced by the DeepSeek 8B results in <xref ref-type="table" rid="table6">Table 6</xref>, where the SE exhibits a substantial reduction. Furthermore, the verbosity of model outputs was significantly reduced after fine-tuning, as reflected in improved Jaccard Index scores and overall performance gains. Previous research has similarly shown that fine-tuning on domain-specific data can significantly enhance performance by adjusting model weights to reflect the unique characteristics of the target domain, such as better handling of abbreviations, acronyms, and clinically relevant contexts [<xref ref-type="bibr" rid="ref58">58</xref>]. Moreover, instruction fine-tuning has been shown to improve the model’s zero-shot performance [<xref ref-type="bibr" rid="ref97">97</xref>].</p>
        <fig id="figure5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Case study for extracting the top 10 important medical jargons from BioMistral 7B and fine-tuning BioMistral 7B on the gold-standard dataset. The fine-tuned model shows more robustness than vanilla models. The highlighted jargons are the ones that overlap with the expert-annotated labels, whereas BioMistral 7B shows misalignment erroneous output. After a few steps of fine-tuning, the model’s behavior was significantly adjusted. Additionally, the verbosity of the output has decreased. COPD: chronic obstructive pulmonary disease.</p>
          </caption>
          <graphic xlink:href="ai_v5i1e75561_fig5.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Our results reveal that different string matching metrics yield divergent performance narratives. Closed-source models set the benchmark under relaxed string matching; however, fine-tuned models achieved superior <italic>F</italic><sub>1</sub>-scores and MRR when evaluated using the Jaccard Index. We contend that using a combination of string matching metrics enables a more nuanced assessment of model performance. Specifically, relaxed string matching prioritizes semantic preservation and the recognition of domain-specific terminology, while the Jaccard Index measures exact token-level overlap, thereby penalizing verbosity and rewarding concise outputs.</p>
      </sec>
      <sec>
        <title>Analysis of Augmented Dataset</title>
        <p>The data augmentation process relied solely on few-shot prompting, with string matching–based filtering as the only quality control measure. Consequently, the augmented dataset may be of lower quality than the gold-standard dataset. To evaluate this, we randomly sampled 100 instances from the augmented dataset and had a clinical expert manually annotate them. Agreement between the expert’s annotations and the synthetic labels was measured using the Jaccard Index [<xref ref-type="bibr" rid="ref98">98</xref>], and to measure the ranking agreement, MRR. The scores were 0.255 and 0.585, respectively. The terms extracted showed low Jaccard Distance, which likely contributed to lower performance. However, for MRR, the ranking showed comparable performance.</p>
      </sec>
      <sec>
        <title>Fine-Tuning With Augmented Data</title>
        <p>We explored the impact of data augmentation using various sizes of MIMIC-IV [<xref ref-type="bibr" rid="ref85">85</xref>] discharge notes to enhance the performance of LLMs. Data augmentation simulated diverse scenarios within medical notes, enabling the LLM to generalize better across a broader range of note types. In accordance with fine-tuning, the performance gains from data augmentation were meaningful (<xref rid="figure3" ref-type="fig">Figure 3</xref>). In some models, we found that using augmented data was helpful in improving the performance (Llama 3.1 8B, BioMistral 7B, DeepSeek 8B) under relaxed string matching in the top10 extraction task. Furthermore, all models trained on the maximum dataset size (n=9995) demonstrated improved <italic>F</italic><sub>1</sub>-scores that exceeded those of closed-source models under the Jaccard ˍindex–based string matching, but not under relaxed string matching. These findings align with prior studies [<xref ref-type="bibr" rid="ref97">97</xref>,<xref ref-type="bibr" rid="ref99">99</xref>], which argue that using LLM-generated data can improve model performance on downstream tasks. Additionally, studies have shown that while data augmentation provides benefits, substantial improvements often require a high degree of variation in the augmented data [<xref ref-type="bibr" rid="ref100">100</xref>,<xref ref-type="bibr" rid="ref101">101</xref>].</p>
        <p>However, fine-tuning with an augmented dataset did not outperform fine-tuning with the gold-standard dataset for other models (Mistral 7B, BioMistral 7B, and DeepSeek 8B) when evaluated using Jaccard Index–based string matching (<xref ref-type="table" rid="table5">Table 5</xref>). Since we only used two examples for ICL, this may have affected the overall quality of the augmented dataset [<xref ref-type="bibr" rid="ref102">102</xref>]. Nonetheless, this still leaves a question of why the LLMs’ performance increased in the first place, considering the quality of the augmented dataset. Although data augmentation did not generate the most “accurate” synthetic dataset, it did not generate critical errors that severely harmed the model’s performance. As a result, the augmented data provides weak but structured supervision that remains correlated with expert judgment. Consistent with prior findings in weakly supervised learning, such signals can still guide models toward meaningful patterns, particularly when aggregated across large-scale data. The observed performance improvements across models suggest that the augmented dataset captures clinically relevant structure despite instance-level disagreement, supporting the validity of our approach.</p>
      </sec>
      <sec>
        <title>Impact of Augmented Dataset Size</title>
        <p>The effect of augmented dataset size on model performance varied depending on the string matching metric. Under relaxed string matching, only BioMistral 7B demonstrated a consistent performance improvement with increasing dataset size (<xref rid="figure3" ref-type="fig">Figure 3</xref>A). Conversely, string matching using the Jaccard Index showed that all models achieved higher <italic>F</italic><sub>1</sub>-scores as the dataset size grew (<xref rid="figure3" ref-type="fig">Figure 3</xref>B), indicating metric-dependent performance trends. Our results partially align with findings from Yuan et al [<xref ref-type="bibr" rid="ref103">103</xref>] and Kim et al [<xref ref-type="bibr" rid="ref104">104</xref>], where increasing the size of synthetic datasets significantly enhanced model performance compared to models trained on smaller real-world datasets. For other models such as DeepSeek 8B, the model’s performance oscillated over dataset size in top 5 and top 10 extraction under relaxed string matching, although it ultimately showed the best performance at n=9995 regardless of the metrics.</p>
      </sec>
      <sec>
        <title>Recommendations for Best Practices</title>
        <p>Based on our findings, we categorize the risks of deploying LLMs into two distinct domains: operational risk and clinical risk. (1) Operational risk encompasses failures in prompt adherence or structural inconsistencies where the model’s output deviates from the required format, potentially causing system-level failures. These risks are largely mitigable through rule-based validation and explicit format-checking mechanisms. (2) Clinical risk refers to the generation of factually incorrect information (hallucinations) that could directly compromise patient safety. Our results demonstrate that despite performance improvements across models, clinical risk remains a critical concern. Consequently, we argue that all LLM-generated outputs must undergo human-in-the-loop verification and be subjected to rigorous fact-checking protocols before clinical implementation.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>This study has several limitations. First, we tested only a limited selection of available closed- and open-source LLMs. Specifically, we only tested models with fewer than 10B parameters, which are relatively small open-source LLMs. Second, our gold-standard dataset, being based solely on physician annotations, captures only the clinical perspective, not considering the actual needs of patients. Third, despite the improvements achieved through fine-tuning and data augmentation, the performance of the models still falls short of human annotation. Fourth, the effectiveness of these methods should be validated in real-world settings, such as Hyper-DREAM, which tests the efficacy of the blood pressure management system [<xref ref-type="bibr" rid="ref105">105</xref>].</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>Our study provides a comprehensive evaluation of closed- and open-source LLMs for identifying and prioritizing medical jargon within expert-annotated EHR notes. Strategic interventions, prompting, fine-tuning, and data augmentation notably enhanced the use of open-source models for domain-specific clinical tasks. Fine-tuning on a gold-standard dataset emerged as the most effective overall strategy, yielding the highest performance gains and a substantial reduction in error counts under Jaccard Index–based string matching. While the data augmentation strategy did not consistently outperform fine-tuning, it demonstrated measurable improvements in model-dependent scenarios. Nevertheless, our error analysis underscores that all evaluated models, regardless of architecture or optimization status, retain inherent degrees of both operational and clinical risk.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Prompts used in data augmentation and prompting strategies.</p>
        <media xlink:href="ai_v5i1e75561_app1.docx" xlink:title="DOCX File , 260 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Additional results with mean (SD) of precision, recall.</p>
        <media xlink:href="ai_v5i1e75561_app2.docx" xlink:title="DOCX File , 66 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">BERT</term>
          <def>
            <p>Bidirectional Encoder Representations from Transformers</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">BioNER</term>
          <def>
            <p>biomedical named entity recognition</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">EHR</term>
          <def>
            <p>electronic health record</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">GE</term>
          <def>
            <p>granularity error</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">ICL</term>
          <def>
            <p>in-context learning</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">ME</term>
          <def>
            <p>misalignment error</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">MIMIC-IV</term>
          <def>
            <p>Medical Information Mart for Intensive Care IV</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">MRR</term>
          <def>
            <p>mean reciprocal rank</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb10">RoBERTa</term>
          <def>
            <p>Robustly Optimized BERT Pretraining approach</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb11">SE</term>
          <def>
            <p>spurious error</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb12">UMLS</term>
          <def>
            <p>Unified Medical Language System</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>We greatly value the University of Massachusetts Biomedical Informatics Natural Language Processing (BioNLP) group’s insightful feedback and thoughtful guidance. All the sections were revised through generative AI tools for checking grammatical errors and correcting awkward sentences. The views expressed in this article are those of the authors and do not necessarily reflect the position or policy of the US Department of Veterans Affairs or the US government.</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>The datasets generated and/or analyzed during the current study are not publicly available due to patient privacy and the terms of the data use agreement governing the source clinical notes. The source code will be released here [<xref ref-type="bibr" rid="ref106">106</xref>].</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>The authors declared no financial support was received for this work.</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>Conceptualization, methodology, software, formal analysis, investigation, data curation, visualization, and writing of the original draft: WSJ</p>
        <p>Methodology, software, formal analysis, investigation, data curation, visualization, and writing of the original draft: SS</p>
        <p>Project administration, writing of the original draft, and writing – review and editing: ZY</p>
        <p>Methodology, validation, and writing – review and editing: ZY, HT, and SK</p>
        <p>Conceptualization, supervision, project administration, funding acquisition, and writing – review and editing: HY</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Delbanco</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Walker</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Darer</surname>
              <given-names>JD</given-names>
            </name>
            <name name-style="western">
              <surname>Elmore</surname>
              <given-names>JG</given-names>
            </name>
            <name name-style="western">
              <surname>Feldman</surname>
              <given-names>HJ</given-names>
            </name>
            <name name-style="western">
              <surname>Leveille</surname>
              <given-names>SG</given-names>
            </name>
          </person-group>
          <article-title>Open notes: doctors and patients signing on</article-title>
          <source>Ann Intern Med</source>
          <year>2010</year>
          <month>07</month>
          <day>20</day>
          <volume>153</volume>
          <issue>2</issue>
          <fpage>121</fpage>
          <lpage>5</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.acpjournals.org/doi/10.7326/0003-4819-153-2-201007200-00008?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.7326/0003-4819-153-2-201007200-00008</pub-id>
          <pub-id pub-id-type="medline">20643992</pub-id>
          <pub-id pub-id-type="pii">153/2/121</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="web">
          <article-title>Blue button</article-title>
          <source>Office of the National Coordinator for Health Information Technology</source>
          <year>2024</year>
          <access-date>2026-06-12</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.healthit.gov/patients-families/about-blue-button-movement">https://www.healthit.gov/patients-families/about-blue-button-movement</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Delbanco</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Walker</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Bell</surname>
              <given-names>SK</given-names>
            </name>
            <name name-style="western">
              <surname>Darer</surname>
              <given-names>JD</given-names>
            </name>
            <name name-style="western">
              <surname>Elmore</surname>
              <given-names>JG</given-names>
            </name>
            <name name-style="western">
              <surname>Farag</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Feldman</surname>
              <given-names>HJ</given-names>
            </name>
          </person-group>
          <article-title>Inviting patients to read their doctors' notes: a quasi-experimental study and a look ahead</article-title>
          <source>Ann Intern Med</source>
          <year>2012</year>
          <month>10</month>
          <day>02</day>
          <volume>157</volume>
          <issue>7</issue>
          <fpage>461</fpage>
          <lpage>70</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.acpjournals.org/doi/10.7326/0003-4819-157-7-201210020-00002?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.7326/0003-4819-157-7-201210020-00002</pub-id>
          <pub-id pub-id-type="medline">23027317</pub-id>
          <pub-id pub-id-type="pii">1363511</pub-id>
          <pub-id pub-id-type="pmcid">PMC3908866</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gabay</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>21st Century Cures Act</article-title>
          <source>Hosp Pharm</source>
          <year>2017</year>
          <month>04</month>
          <volume>52</volume>
          <issue>4</issue>
          <fpage>264</fpage>
          <lpage>265</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/28515504"/>
          </comment>
          <pub-id pub-id-type="doi">10.1310/hpj5204-264</pub-id>
          <pub-id pub-id-type="medline">28515504</pub-id>
          <pub-id pub-id-type="pmcid">PMC5424829</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bajwa</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Munir</surname>
              <given-names>U</given-names>
            </name>
            <name name-style="western">
              <surname>Nori</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Williams</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Artificial intelligence in healthcare: transforming the practice of medicine</article-title>
          <source>Future Healthc J</source>
          <year>2021</year>
          <volume>8</volume>
          <issue>2</issue>
          <fpage>e188</fpage>
          <lpage>e194</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2514-6645(24)00527-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.7861/fhj.2021-0095</pub-id>
          <pub-id pub-id-type="medline">34286183</pub-id>
          <pub-id pub-id-type="pii">S2514-6645(24)00527-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC8285156</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lye</surname>
              <given-names>CT</given-names>
            </name>
            <name name-style="western">
              <surname>Forman</surname>
              <given-names>HP</given-names>
            </name>
            <name name-style="western">
              <surname>Daniel</surname>
              <given-names>JG</given-names>
            </name>
            <name name-style="western">
              <surname>Krumholz</surname>
              <given-names>HM</given-names>
            </name>
          </person-group>
          <article-title>The 21st Century Cures Act and electronic health records one year later: will patients see the benefits?</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2018</year>
          <volume>25</volume>
          <issue>9</issue>
          <fpage>1218</fpage>
          <lpage>1220</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/30184156"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocy065</pub-id>
          <pub-id pub-id-type="medline">30184156</pub-id>
          <pub-id pub-id-type="pii">5060211</pub-id>
          <pub-id pub-id-type="pmcid">PMC7646899</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Arvisais-Anhalt</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lau</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Lehmann</surname>
              <given-names>CU</given-names>
            </name>
            <name name-style="western">
              <surname>Holmgren</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Medford</surname>
              <given-names>RJ</given-names>
            </name>
            <name name-style="western">
              <surname>Ramirez</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>CN</given-names>
            </name>
          </person-group>
          <article-title>The 21st Century Cures Act and multiuser electronic health record access: potential pitfalls of information release</article-title>
          <source>J Med Internet Res</source>
          <year>2022</year>
          <month>02</month>
          <day>17</day>
          <volume>24</volume>
          <issue>2</issue>
          <fpage>e34085</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2022/2/e34085/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/34085</pub-id>
          <pub-id pub-id-type="medline">35175207</pub-id>
          <pub-id pub-id-type="pii">v24i2e34085</pub-id>
          <pub-id pub-id-type="pmcid">PMC8895284</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rodriguez</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Clark</surname>
              <given-names>CR</given-names>
            </name>
            <name name-style="western">
              <surname>Bates</surname>
              <given-names>DW</given-names>
            </name>
          </person-group>
          <article-title>Digital health equity as a necessity in the 21st Century Cures Act era</article-title>
          <source>JAMA</source>
          <year>2020</year>
          <volume>323</volume>
          <issue>23</issue>
          <fpage>2381</fpage>
          <lpage>2382</lpage>
          <pub-id pub-id-type="doi">10.1001/jama.2020.7858</pub-id>
          <pub-id pub-id-type="medline">32463421</pub-id>
          <pub-id pub-id-type="pii">2766776</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nutbeam</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Artificial intelligence and health literacy—proceed with caution</article-title>
          <source>Health Literacy and Communication Open</source>
          <year>2023</year>
          <month>10</month>
          <day>17</day>
          <volume>1</volume>
          <issue>1</issue>
          <fpage>2263355</fpage>
          <pub-id pub-id-type="doi">10.1080/28355245.2023.2263355</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Root</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Oster</surname>
              <given-names>NV</given-names>
            </name>
            <name name-style="western">
              <surname>Jackson</surname>
              <given-names>SL</given-names>
            </name>
            <name name-style="western">
              <surname>Mejilla</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Walker</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Elmore</surname>
              <given-names>JG</given-names>
            </name>
          </person-group>
          <article-title>Characteristics of patients who report confusion after reading their primary care clinic notes online</article-title>
          <source>Health Commun</source>
          <year>2016</year>
          <volume>31</volume>
          <issue>6</issue>
          <fpage>778</fpage>
          <lpage>781</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/26529325"/>
          </comment>
          <pub-id pub-id-type="doi">10.1080/10410236.2014.990078</pub-id>
          <pub-id pub-id-type="medline">26529325</pub-id>
          <pub-id pub-id-type="pmcid">PMC7043205</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kayastha</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Pollak</surname>
              <given-names>KI</given-names>
            </name>
            <name name-style="western">
              <surname>LeBlanc</surname>
              <given-names>TW</given-names>
            </name>
          </person-group>
          <article-title>Open oncology notes: a qualitative study of oncology patients’ experiences reading their cancer care notes</article-title>
          <source>JOP</source>
          <year>2018</year>
          <month>04</month>
          <volume>14</volume>
          <issue>4</issue>
          <fpage>e251</fpage>
          <lpage>e258</lpage>
          <pub-id pub-id-type="doi">10.1200/jop.2017.028605</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kujala</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hörhammer</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Väyrynen</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Holmroos</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Nättiaho-Rönnholm</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hägglund</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Johansen</surname>
              <given-names>MA</given-names>
            </name>
          </person-group>
          <article-title>Patients' experiences of web-based access to electronic health records in Finland: cross-sectional survey</article-title>
          <source>J Med Internet Res</source>
          <year>2022</year>
          <month>06</month>
          <day>06</day>
          <volume>24</volume>
          <issue>6</issue>
          <fpage>e37438</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2022/6/e37438/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/37438</pub-id>
          <pub-id pub-id-type="medline">35666563</pub-id>
          <pub-id pub-id-type="pii">v24i6e37438</pub-id>
          <pub-id pub-id-type="pmcid">PMC9210208</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Choudhry</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Baghdadi</surname>
              <given-names>YM</given-names>
            </name>
            <name name-style="western">
              <surname>Wagie</surname>
              <given-names>AE</given-names>
            </name>
            <name name-style="western">
              <surname>Habermann</surname>
              <given-names>EB</given-names>
            </name>
            <name name-style="western">
              <surname>Heller</surname>
              <given-names>SF</given-names>
            </name>
            <name name-style="western">
              <surname>Jenkins</surname>
              <given-names>DH</given-names>
            </name>
            <name name-style="western">
              <surname>Cullinane</surname>
              <given-names>DC</given-names>
            </name>
          </person-group>
          <article-title>Readability of discharge summaries: with what level of information are we dismissing our patients?</article-title>
          <source>Am J Surg</source>
          <year>2016</year>
          <month>03</month>
          <volume>211</volume>
          <issue>3</issue>
          <fpage>631</fpage>
          <lpage>6</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/26794665"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.amjsurg.2015.12.005</pub-id>
          <pub-id pub-id-type="medline">26794665</pub-id>
          <pub-id pub-id-type="pii">S0002-9610(15)30040-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC5245984</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Khasawneh</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kratzke</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Adapa</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Marks</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Mazur</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Effect of notes' access and complexity on OpenNotes' utility</article-title>
          <source>Appl Clin Inform</source>
          <year>2022</year>
          <month>10</month>
          <volume>13</volume>
          <issue>5</issue>
          <fpage>1015</fpage>
          <lpage>1023</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://www.thieme-connect.com/DOI/DOI?10.1055/a-1942-6889"/>
          </comment>
          <pub-id pub-id-type="doi">10.1055/a-1942-6889</pub-id>
          <pub-id pub-id-type="medline">36104159</pub-id>
          <pub-id pub-id-type="pmcid">PMC9605819</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rahimian</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Warner</surname>
              <given-names>JL</given-names>
            </name>
            <name name-style="western">
              <surname>Salmi</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Rosenbloom</surname>
              <given-names>ST</given-names>
            </name>
            <name name-style="western">
              <surname>Davis</surname>
              <given-names>RB</given-names>
            </name>
            <name name-style="western">
              <surname>Joyce</surname>
              <given-names>RM</given-names>
            </name>
          </person-group>
          <article-title>Open notes sounds great, but will a provider's documentation change? An exploratory study of the effect of open notes on oncology documentation</article-title>
          <source>JAMIA Open</source>
          <year>2021</year>
          <volume>4</volume>
          <issue>3</issue>
          <fpage>ooab051</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/34661067"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamiaopen/ooab051</pub-id>
          <pub-id pub-id-type="medline">34661067</pub-id>
          <pub-id pub-id-type="pii">ooab051</pub-id>
          <pub-id pub-id-type="pmcid">PMC8518311</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Readability formulas and user perceptions of electronic health records difficulty: a corpus study</article-title>
          <source>J Med Internet Res</source>
          <year>2017</year>
          <month>03</month>
          <day>02</day>
          <volume>19</volume>
          <issue>3</issue>
          <fpage>e59</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2017/3/e59/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/jmir.6962</pub-id>
          <pub-id pub-id-type="medline">28254738</pub-id>
          <pub-id pub-id-type="pii">v19i3e59</pub-id>
          <pub-id pub-id-type="pmcid">PMC5355629</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zeng-Treitler</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Goryachev</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Keselman</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Slaughter</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Smith</surname>
              <given-names>CA</given-names>
            </name>
          </person-group>
          <article-title>Text characteristics of clinical reports and their implications for the readability of personal health records</article-title>
          <source>Stud Health Technol Inform</source>
          <year>2007</year>
          <volume>129</volume>
          <issue>Pt 2</issue>
          <fpage>1117</fpage>
          <lpage>1121</lpage>
          <pub-id pub-id-type="medline">17911889</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Polepalli</surname>
              <given-names>RB</given-names>
            </name>
            <name name-style="western">
              <surname>Houston</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Brandt</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Fang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Improving patients' electronic health record comprehension with NoteAid</article-title>
          <source>Medinfo 2013 IOS Press</source>
          <year>2013</year>
          <fpage>714</fpage>
          <lpage>718</lpage>
          <pub-id pub-id-type="doi">10.3233/978-1-61499-289-9-714</pub-id>
          <pub-id pub-id-type="medline">23920650</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sarzynski</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Hashmi</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Subramanian</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Fitzpatrick</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Polverento</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Simmons</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Brooks</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Opportunities to improve clinical summaries for patients at hospital discharge</article-title>
          <source>BMJ Qual Saf</source>
          <year>2017</year>
          <month>05</month>
          <volume>26</volume>
          <issue>5</issue>
          <fpage>372</fpage>
          <lpage>380</lpage>
          <pub-id pub-id-type="doi">10.1136/bmjqs-2015-005201</pub-id>
          <pub-id pub-id-type="medline">27154878</pub-id>
          <pub-id pub-id-type="pii">bmjqs-2015-005201</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Doak</surname>
              <given-names>CC</given-names>
            </name>
            <name name-style="western">
              <surname>Doak</surname>
              <given-names>LG</given-names>
            </name>
            <name name-style="western">
              <surname>Root</surname>
              <given-names>JH</given-names>
            </name>
          </person-group>
          <article-title>Teaching patients with low literacy skills</article-title>
          <source>American Journal of Nursing</source>
          <year>1996</year>
          <volume>96</volume>
          <issue>12</issue>
          <fpage>16M</fpage>
          <pub-id pub-id-type="doi">10.1097/00000446-199612000-00022</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Doak</surname>
              <given-names>CC</given-names>
            </name>
            <name name-style="western">
              <surname>Doak</surname>
              <given-names>LG</given-names>
            </name>
            <name name-style="western">
              <surname>Friedell</surname>
              <given-names>GH</given-names>
            </name>
            <name name-style="western">
              <surname>Meade</surname>
              <given-names>CD</given-names>
            </name>
          </person-group>
          <article-title>Improving comprehension for cancer patients with low literacy skills: strategies for clinicians</article-title>
          <source>CA Cancer J Clin</source>
          <year>1998</year>
          <volume>48</volume>
          <issue>3</issue>
          <fpage>151</fpage>
          <lpage>62</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://onlinelibrary.wiley.com/resolve/openurl?genre=article&#38;sid=nlm:pubmed&#38;issn=0007-9235&#38;date=1998&#38;volume=48&#38;issue=3&#38;spage=151"/>
          </comment>
          <pub-id pub-id-type="doi">10.3322/canjclin.48.3.151</pub-id>
          <pub-id pub-id-type="medline">9594918</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Walsh</surname>
              <given-names>TM</given-names>
            </name>
            <name name-style="western">
              <surname>Volsko</surname>
              <given-names>TA</given-names>
            </name>
          </person-group>
          <article-title>Readability assessment of internet-based consumer health information</article-title>
          <source>Respir Care</source>
          <year>2008</year>
          <month>10</month>
          <volume>53</volume>
          <issue>10</issue>
          <fpage>1310</fpage>
          <lpage>5</lpage>
          <pub-id pub-id-type="medline">18811992</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Eltorai</surname>
              <given-names>AE</given-names>
            </name>
            <name name-style="western">
              <surname>Han</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Truntzer</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Daniels</surname>
              <given-names>AH</given-names>
            </name>
          </person-group>
          <article-title>Readability of patient education materials on the American Orthopaedic Society for Sports Medicine website</article-title>
          <source>Phys Sportsmed</source>
          <year>2014</year>
          <volume>42</volume>
          <issue>4</issue>
          <fpage>125</fpage>
          <lpage>130</lpage>
          <pub-id pub-id-type="doi">10.3810/psm.2014.11.2099</pub-id>
          <pub-id pub-id-type="medline">25419896</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Morony</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Flynn</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>McCaffery</surname>
              <given-names>KJ</given-names>
            </name>
            <name name-style="western">
              <surname>Jansen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>AC</given-names>
            </name>
          </person-group>
          <article-title>Readability of written materials for CKD patients: a systematic review</article-title>
          <source>Am J Kidney Dis</source>
          <year>2015</year>
          <month>06</month>
          <volume>65</volume>
          <issue>6</issue>
          <fpage>842</fpage>
          <lpage>50</lpage>
          <pub-id pub-id-type="doi">10.1053/j.ajkd.2014.11.025</pub-id>
          <pub-id pub-id-type="medline">25661679</pub-id>
          <pub-id pub-id-type="pii">S0272-6386(14)01535-2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>SB</given-names>
            </name>
            <name name-style="western">
              <surname>Farach</surname>
              <given-names>FJ</given-names>
            </name>
            <name name-style="western">
              <surname>Pelphrey</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Rozenblit</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Data management in clinical research: synthesizing stakeholder perspectives</article-title>
          <source>J Biomed Inform</source>
          <year>2016</year>
          <month>04</month>
          <volume>60</volume>
          <fpage>286</fpage>
          <lpage>93</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(16)00036-8"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2016.02.014</pub-id>
          <pub-id pub-id-type="medline">26925516</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(16)00036-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Morid</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Fiszman</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Raja</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Jonnalagadda</surname>
              <given-names>SR</given-names>
            </name>
            <name name-style="western">
              <surname>Del Fiol</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Classification of clinically useful sentences in clinical evidence resources</article-title>
          <source>J Biomed Inform</source>
          <year>2016</year>
          <month>04</month>
          <volume>60</volume>
          <fpage>14</fpage>
          <lpage>22</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(16)00004-6"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2016.01.003</pub-id>
          <pub-id pub-id-type="medline">26774763</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(16)00004-6</pub-id>
          <pub-id pub-id-type="pmcid">PMC4836984</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kandula</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Curtis</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Zeng-Treitler</surname>
              <given-names>Q</given-names>
            </name>
          </person-group>
          <article-title>A semantic and syntactic text simplification tool for health content</article-title>
          <source>AMIA Annu Symp Proc</source>
          <year>2010</year>
          <month>11</month>
          <day>13</day>
          <volume>2010</volume>
          <fpage>366</fpage>
          <lpage>70</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/21347002"/>
          </comment>
          <pub-id pub-id-type="medline">21347002</pub-id>
          <pub-id pub-id-type="pmcid">PMC3041424</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zeng-Treitler</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Goryachev</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Keselman</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Rosendale</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Making texts in electronic health records comprehensible to consumers: a prototype translator</article-title>
          <source>AMIA Annu Symp Proc</source>
          <year>2007</year>
          <month>10</month>
          <day>11</day>
          <volume>2007</volume>
          <fpage>846</fpage>
          <lpage>50</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/18693956"/>
          </comment>
          <pub-id pub-id-type="medline">18693956</pub-id>
          <pub-id pub-id-type="pmcid">PMC2655860</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Abrahamsson</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Forni</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Skeppstedt</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kvist</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Medical text simplification using synonym replacement: adapting assessment of word difficulty to a compounding language</article-title>
          <year>2014</year>
          <conf-name>Proceedings of the 3rd Workshop on Predicting and Improving text Readability for Target Reader Populations (PITR)</conf-name>
          <conf-date>April 10, 2014</conf-date>
          <conf-loc>Gothenburg, Sweden</conf-loc>
          <fpage>57</fpage>
          <lpage>65</lpage>
          <pub-id pub-id-type="doi">10.3115/v1/w14-1207</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Methods for linking EHR notes to education materials</article-title>
          <source>Inf Retrieval J</source>
          <year>2015</year>
          <month>09</month>
          <day>03</day>
          <volume>19</volume>
          <issue>1-2</issue>
          <fpage>174</fpage>
          <lpage>188</lpage>
          <pub-id pub-id-type="doi">10.1007/s10791-015-9263-1</pub-id>
          <pub-id pub-id-type="medline">26306273</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Druhl</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Polepalli Ramesh</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Houston</surname>
              <given-names>TK</given-names>
            </name>
            <name name-style="western">
              <surname>Brandt</surname>
              <given-names>CA</given-names>
            </name>
            <name name-style="western">
              <surname>Zulman</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Vimalananda</surname>
              <given-names>VG</given-names>
            </name>
          </person-group>
          <article-title>A natural language processing system that links medical terms in electronic health record notes to lay definitions: system development using physician reviews</article-title>
          <source>J Med Internet Res</source>
          <year>2018</year>
          <month>01</month>
          <day>22</day>
          <volume>20</volume>
          <issue>1</issue>
          <fpage>e26</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2018/1/e26/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/jmir.8669</pub-id>
          <pub-id pub-id-type="medline">29358159</pub-id>
          <pub-id pub-id-type="pii">v20i1e26</pub-id>
          <pub-id pub-id-type="pmcid">PMC5799720</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kwon</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Jordan</surname>
              <given-names>HS</given-names>
            </name>
            <name name-style="western">
              <surname>Levy</surname>
              <given-names>DA</given-names>
            </name>
            <name name-style="western">
              <surname>Corner</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>MedJEx: a medical jargon extraction model with Wiki's hyperlink sspan and contextualized masked language model score</article-title>
          <source>Proc Conf Empir Methods Nat Lang Process</source>
          <year>2022</year>
          <month>12</month>
          <volume>2022</volume>
          <fpage>11733</fpage>
          <lpage>11751</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37103473"/>
          </comment>
          <pub-id pub-id-type="medline">37103473</pub-id>
          <pub-id pub-id-type="pmcid">PMC10129059</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Leroy</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Endicott</surname>
              <given-names>JE</given-names>
            </name>
            <name name-style="western">
              <surname>Mouradi</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Kauchak</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Just</surname>
              <given-names>ML</given-names>
            </name>
          </person-group>
          <article-title>Improving perceived and actual text difficulty for health information consumers using semi-automated methods</article-title>
          <source>AMIA Annu Symp Proc</source>
          <year>2012</year>
          <volume>2012</volume>
          <fpage>522</fpage>
          <lpage>31</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/23304324"/>
          </comment>
          <pub-id pub-id-type="medline">23304324</pub-id>
          <pub-id pub-id-type="pmcid">PMC3540563</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Finding important terms for patients in their electronic health records: a learning-to-rank approach using expert annotations</article-title>
          <source>JMIR Med Inform</source>
          <year>2016</year>
          <month>11</month>
          <day>30</day>
          <volume>4</volume>
          <issue>4</issue>
          <fpage>e40</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://medinform.jmir.org/2016/4/e40/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/medinform.6373</pub-id>
          <pub-id pub-id-type="medline">27903489</pub-id>
          <pub-id pub-id-type="pii">v4i4e40</pub-id>
          <pub-id pub-id-type="pmcid">PMC5156821</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Unsupervised ensemble ranking of terms in electronic health record notes based on their importance to patients</article-title>
          <source>J Biomed Inform</source>
          <year>2017</year>
          <month>04</month>
          <volume>68</volume>
          <fpage>121</fpage>
          <lpage>131</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(17)30045-X"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2017.02.016</pub-id>
          <pub-id pub-id-type="medline">28267590</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(17)30045-X</pub-id>
          <pub-id pub-id-type="pmcid">PMC5505865</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Aronson</surname>
              <given-names>AR</given-names>
            </name>
          </person-group>
          <article-title>Effective mapping of biomedical text to the UMLS Metathesaurus: the MetaMap program</article-title>
          <source>Proc AMIA Symp</source>
          <year>2001</year>
          <fpage>17</fpage>
          <lpage>21</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/11825149"/>
          </comment>
          <pub-id pub-id-type="medline">11825149</pub-id>
          <pub-id pub-id-type="pii">D010001275</pub-id>
          <pub-id pub-id-type="pmcid">PMC2243666</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Neumann</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>King</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Beltagy</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Ammar</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>ScispaCy: fast and robust models for biomedical natural language processing</article-title>
          <source>Association for Computational Linguistics</source>
          <year>2019</year>
          <conf-name>Proceedings of the 18th BioNLP Workshop and Shared Task</conf-name>
          <conf-date>August 1, 2019</conf-date>
          <conf-loc>Florence, Italy</conf-loc>
          <fpage>190207669</fpage>
          <pub-id pub-id-type="doi">10.18653/v1/w19-5034</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Eyre</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Chapman</surname>
              <given-names>AB</given-names>
            </name>
            <name name-style="western">
              <surname>Peterson</surname>
              <given-names>KS</given-names>
            </name>
            <name name-style="western">
              <surname>Shi</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Alba</surname>
              <given-names>PR</given-names>
            </name>
            <name name-style="western">
              <surname>Jones</surname>
              <given-names>MM</given-names>
            </name>
            <name name-style="western">
              <surname>Box</surname>
              <given-names>TL</given-names>
            </name>
          </person-group>
          <article-title>Launching into clinical space with medspaCy: a new clinical text processing toolkit in Python</article-title>
          <source>AMIA Annu Symp Proc</source>
          <year>2021</year>
          <volume>2021</volume>
          <fpage>438</fpage>
          <lpage>447</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/35308962"/>
          </comment>
          <pub-id pub-id-type="medline">35308962</pub-id>
          <pub-id pub-id-type="pii">3576697</pub-id>
          <pub-id pub-id-type="pmcid">PMC8861690</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref39">
        <label>39</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Soldaini</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Goharian</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>QuickUMLS: a fast, unsupervised approach for medical concept extraction</article-title>
          <year>2016</year>
          <conf-name>SIGIR MedIR Workshop</conf-name>
          <conf-date>July 21, 2016</conf-date>
          <conf-loc>Pisa, Italy</conf-loc>
          <fpage>1</fpage>
          <lpage>4</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://medir2016.imag.fr/data/MEDIR_2016_paper_16.pdf"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref40">
        <label>40</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bodenreider</surname>
              <given-names>O</given-names>
            </name>
          </person-group>
          <article-title>The Unified Medical Language System (UMLS): integrating biomedical terminology</article-title>
          <source>Nucleic Acids Res</source>
          <year>2004</year>
          <month>01</month>
          <day>01</day>
          <volume>32</volume>
          <issue>Database issue</issue>
          <fpage>D267</fpage>
          <lpage>70</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/14681409"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/nar/gkh061</pub-id>
          <pub-id pub-id-type="medline">14681409</pub-id>
          <pub-id pub-id-type="pii">32/suppl_1/D267</pub-id>
          <pub-id pub-id-type="pmcid">PMC308795</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref41">
        <label>41</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chung</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Scales</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>Large language models encode clinical knowledge</article-title>
          <source>Nature</source>
          <year>2023</year>
          <month>08</month>
          <volume>620</volume>
          <issue>7972</issue>
          <fpage>172</fpage>
          <lpage>180</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37438534"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="medline">37438534</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC10396962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref42">
        <label>42</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tian</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Yeganova</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Lai</surname>
              <given-names>PT</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Opportunities and challenges for ChatGPT and large language models in biomedicine and health</article-title>
          <source>Brief Bioinform</source>
          <year>2023</year>
          <month>11</month>
          <day>22</day>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>bbad493</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38168838"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bib/bbad493</pub-id>
          <pub-id pub-id-type="medline">38168838</pub-id>
          <pub-id pub-id-type="pii">7505071</pub-id>
          <pub-id pub-id-type="pmcid">PMC10762511</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref43">
        <label>43</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Palepu</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schaekermann</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Saab</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Freyberg</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tanno</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Towards conversational diagnostic AI</article-title>
          <source>ArXiv. Preprint posted online on January 11, 2024</source>
          <year>2024</year>
          <month>02</month>
          <day>22</day>
          <volume>1</volume>
          <issue>3</issue>
          <fpage>46</fpage>
          <pub-id pub-id-type="doi">10.1056/AIoa2300138</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref44">
        <label>44</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>McDuff</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Schaekermann</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Palepu</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Garrison</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Towards accurate differential diagnosis with large language models</article-title>
          <source>Nature</source>
          <year>2025</year>
          <month>06</month>
          <volume>642</volume>
          <issue>8067</issue>
          <fpage>451</fpage>
          <lpage>457</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41586-025-08869-4"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-025-08869-4</pub-id>
          <pub-id pub-id-type="medline">40205049</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-025-08869-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC12158753</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref45">
        <label>45</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>PMC-LLaMA: toward building open-source language models for medicine</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2024</year>
          <month>09</month>
          <day>01</day>
          <volume>31</volume>
          <issue>9</issue>
          <fpage>1833</fpage>
          <lpage>1843</lpage>
          <pub-id pub-id-type="doi">10.1093/jamia/ocae045</pub-id>
          <pub-id pub-id-type="medline">38613821</pub-id>
          <pub-id pub-id-type="pii">7645318</pub-id>
          <pub-id pub-id-type="pmcid">PMC11639126</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref46">
        <label>46</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Cano</surname>
              <given-names>AH</given-names>
            </name>
            <name name-style="western">
              <surname>Romanou</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Bonnet</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Matoba</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Salvi</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Pagliardini</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Meditron-70b: scaling medical pretraining for large language models</article-title>
          <source>ArXiv. Preprint posted online on November 27, 2023</source>
          <year>2023</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2311.16079"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2311.16079</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref47">
        <label>47</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tran</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>BioInstruct: instruction tuning of large language models for biomedical natural language processing</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2024</year>
          <month>09</month>
          <day>01</day>
          <volume>31</volume>
          <issue>9</issue>
          <fpage>1821</fpage>
          <lpage>1832</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/jamia/article-lookup/doi/10.1093/jamia/ocae122"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocae122</pub-id>
          <pub-id pub-id-type="medline">38833265</pub-id>
          <pub-id pub-id-type="pii">7687618</pub-id>
          <pub-id pub-id-type="pmcid">PMC11339494</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref48">
        <label>48</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nori</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>King</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>McKinney</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Carignan</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Horvitz</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Capabilities of GPT-4 on medical challenge problems</article-title>
          <source>arXiv</source>
          <year>2023</year>
          <pub-id pub-id-type="doi">10.5260/chara.21.2.8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref49">
        <label>49</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kung</surname>
              <given-names>TH</given-names>
            </name>
            <name name-style="western">
              <surname>Cheatham</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Medenilla</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sillos</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>De Leon</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Elepaño</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Madriaga</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models</article-title>
          <source>PLOS Digit Health</source>
          <year>2023</year>
          <month>02</month>
          <volume>2</volume>
          <issue>2</issue>
          <fpage>e0000198</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pdig.0000198"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pdig.0000198</pub-id>
          <pub-id pub-id-type="medline">36812645</pub-id>
          <pub-id pub-id-type="pii">PDIG-D-22-00371</pub-id>
          <pub-id pub-id-type="pmcid">PMC9931230</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref50">
        <label>50</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <collab>Google Research</collab>
            <collab>Google DeepMind</collab>
          </person-group>
          <article-title>Advancing multimodal medical capabilities of Gemini</article-title>
          <source>ArXiv. Preprint posted online on May 6, 2024</source>
          <year>2024</year>
          <fpage>240503162</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2405.03162"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2405.03162</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref51">
        <label>51</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Tasmin</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Vashisht</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Jang</surname>
              <given-names>WS</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Berlowitz</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Performance of multimodal GPT-4V on USMLE with image: potential for imaging diagnostic support with explanations</article-title>
          <source>medRxiv. Preprint posted online on November 15, 2023</source>
          <year>2023</year>
          <fpage>10</fpage>
          <pub-id pub-id-type="doi">10.1101/2023.10.26.23297629</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref52">
        <label>52</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Bian</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Medqa-cs: benchmarking large language models clinical skills using an AI-SCE framework</article-title>
          <source>ArXiv. Preprint posted online on January 18, 2026</source>
          <year>2024</year>
          <fpage>74</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2410.01553"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref53">
        <label>53</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Du</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Keloth</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Zuo</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Improving large language models for clinical named entity recognition via prompt engineering</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2024</year>
          <month>09</month>
          <day>01</day>
          <volume>31</volume>
          <issue>9</issue>
          <fpage>1812</fpage>
          <lpage>1820</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/jamia/article-lookup/doi/10.1093/jamia/ocad259"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocad259</pub-id>
          <pub-id pub-id-type="medline">38281112</pub-id>
          <pub-id pub-id-type="pii">7590607</pub-id>
          <pub-id pub-id-type="pmcid">PMC11339492</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref54">
        <label>54</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Monajatipoor</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Stremmel</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Emami</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Mohaghegh</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Rouhsedaghat</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>LLMs in biomedicine: a study on clinical named entity recognition</article-title>
          <source>ArXiv. Preprint posted online on July 11, 2024</source>
          <year>2024</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2404.07376"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2404.07376</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref55">
        <label>55</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>Zero-shot information extraction from radiological reports using ChatGPT</article-title>
          <source>Int J Med Inform</source>
          <year>2024</year>
          <month>03</month>
          <volume>183</volume>
          <fpage>105321</fpage>
          <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2023.105321</pub-id>
          <pub-id pub-id-type="medline">38157785</pub-id>
          <pub-id pub-id-type="pii">S1386-5056(23)00339-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref56">
        <label>56</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Xiu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhong</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Evaluating medical entity recognition in health care: entity model quantitative study</article-title>
          <source>JMIR Med Inform</source>
          <year>2024</year>
          <volume>12</volume>
          <issue>1</issue>
          <fpage>e59782</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://medinform.jmir.org/2024//e59782/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/59782</pub-id>
          <pub-id pub-id-type="medline">39419501</pub-id>
          <pub-id pub-id-type="pii">v12i1e59782</pub-id>
          <pub-id pub-id-type="pmcid">PMC11528166</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref57">
        <label>57</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bose</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Srinivasan</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sleeman</surname>
              <given-names>WC</given-names>
            </name>
            <name name-style="western">
              <surname>Palta</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kapoor</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ghosh</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>A survey on recent named entity recognition and relationship extraction techniques on clinical texts</article-title>
          <source>Applied Sciences</source>
          <year>2021</year>
          <volume>11</volume>
          <issue>18</issue>
          <fpage>8319</fpage>
          <pub-id pub-id-type="doi">10.3390/app11188319</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref58">
        <label>58</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yoon</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>So</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Kang</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>BioBERT: a pre-trained biomedical language representation model for biomedical text mining</article-title>
          <source>Bioinformatics</source>
          <year>2020</year>
          <month>02</month>
          <day>15</day>
          <volume>36</volume>
          <issue>4</issue>
          <fpage>1234</fpage>
          <lpage>1240</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/31501885"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id>
          <pub-id pub-id-type="medline">31501885</pub-id>
          <pub-id pub-id-type="pii">5566506</pub-id>
          <pub-id pub-id-type="pmcid">PMC7703786</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref59">
        <label>59</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ott</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Goyal</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Du</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Joshi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Roberta: A robustly optimized bert pretraining approach</article-title>
          <source>ArXiv. Preprint posted online on July 26, 2019</source>
          <year>2019</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/1907.11692"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref60">
        <label>60</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Cao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Deshpande</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Extracting biomedical factual knowledge using pretrained language model and electronic health record context</article-title>
          <source>AMIA Annu Symp Proc</source>
          <year>2022</year>
          <volume>2022</volume>
          <fpage>1188</fpage>
          <lpage>1197</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37128373"/>
          </comment>
          <pub-id pub-id-type="medline">37128373</pub-id>
          <pub-id pub-id-type="pii">1077</pub-id>
          <pub-id pub-id-type="pmcid">PMC10148358</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref61">
        <label>61</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Cao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Context variance evaluation of pretrained language models for prompt-based biomedical knowledge probing</article-title>
          <source>AMIA Jt Summits Transl Sci Proc</source>
          <year>2023</year>
          <volume>2023</volume>
          <fpage>592</fpage>
          <lpage>601</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37350903"/>
          </comment>
          <pub-id pub-id-type="medline">37350903</pub-id>
          <pub-id pub-id-type="pii">2079</pub-id>
          <pub-id pub-id-type="pmcid">PMC10283095</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref62">
        <label>62</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gutierrez</surname>
              <given-names>BJ</given-names>
            </name>
            <name name-style="western">
              <surname>McNeal</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Washington</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Thinking about GPT-3 in-context learning for biomedical IE' think again</article-title>
          <source>Association for Computational Linguistics</source>
          <year>2022</year>
          <fpage>4497</fpage>
          <lpage>4512</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/2022.findings-emnlp.329</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref63">
        <label>63</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Moradi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Blagec</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Haberl</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Samwald</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Gpt-3 models are poor few-shot learners in the biomedical domain</article-title>
          <source>ArXiv. Preprint posted online on September 6, 2021</source>
          <year>2021</year>
          <fpage>210902555</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2109.02555"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref64">
        <label>64</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Devlin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>MW</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Toutanova</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>BERT: Pre-training of deep bidirectional transformers for language understanding</article-title>
          <source>ArXiv. Preprint posted online on October 11, 2018</source>
          <year>2018</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/1810.04805"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref65">
        <label>65</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Alsentzer</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Murphy</surname>
              <given-names>JR</given-names>
            </name>
            <name name-style="western">
              <surname>Boag</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Weng</surname>
              <given-names>WH</given-names>
            </name>
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Naumann</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>McDermott</surname>
              <given-names>MBA</given-names>
            </name>
          </person-group>
          <article-title>Publicly available clinical BERT embeddings</article-title>
          <source>ArXiv. Preprint posted online on April 6, 2019</source>
          <year>2019</year>
          <month>4</month>
          <day>6</day>
          <fpage>1</fpage>
          <lpage>7</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://arxiv.org/abs/1904.03323"/>
          </comment>
          <pub-id pub-id-type="doi">10.18653/v1/w19-1909</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref66">
        <label>66</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lim</surname>
              <given-names>JH</given-names>
            </name>
            <name name-style="western">
              <surname>Kwon</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Lalor</surname>
              <given-names>JP</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Large language model-based role-playing for personalized medical jargon extraction</article-title>
          <source>ArXiv. Preprint posted online on August 10, 2024</source>
          <year>2024</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2408.05555"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref67">
        <label>67</label>
        <nlm-citation citation-type="web">
          <source>OpenAI</source>
          <access-date>2026-06-12</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://openai.com/">https://openai.com/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref68">
        <label>68</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ghali</surname>
              <given-names>MK</given-names>
            </name>
            <name name-style="western">
              <surname>Farrag</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sakai</surname>
              <given-names>HE</given-names>
            </name>
            <name name-style="western">
              <surname>Baz</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Lam</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Gamedx: generative AI-based medical entity data extractor using large language models</article-title>
          <source>ArXiv. Preprint posted online on May 31, 2024</source>
          <year>2024</year>
          <fpage>240520585</fpage>
          <pub-id pub-id-type="doi">10.2139/ssrn.5063216</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref69">
        <label>69</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Butler</surname>
              <given-names>JJ</given-names>
            </name>
            <name name-style="western">
              <surname>Harrington</surname>
              <given-names>MC</given-names>
            </name>
            <name name-style="western">
              <surname>Tong</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Rosenbaum</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Samsonov</surname>
              <given-names>AP</given-names>
            </name>
            <name name-style="western">
              <surname>Walls</surname>
              <given-names>RJ</given-names>
            </name>
            <name name-style="western">
              <surname>Kennedy</surname>
              <given-names>JG</given-names>
            </name>
          </person-group>
          <article-title>From jargon to clarity: improving the readability of foot and ankle radiology reports with an artificial intelligence large language model</article-title>
          <source>Foot Ankle Surg</source>
          <year>2024</year>
          <month>06</month>
          <volume>30</volume>
          <issue>4</issue>
          <fpage>331</fpage>
          <lpage>337</lpage>
          <pub-id pub-id-type="doi">10.1016/j.fas.2024.01.008</pub-id>
          <pub-id pub-id-type="medline">38336501</pub-id>
          <pub-id pub-id-type="pii">S1268-7731(24)00026-2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref70">
        <label>70</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mannhardt</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Bondi-Kelly</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Lam</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>O'Connell</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Mozannar</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Asiedu</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Mozannar</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Impact of large language model assistance on patients reading clinical notes: a mixed-methods study</article-title>
          <source>ArXiv. Preprint posted online on January 17, 2024</source>
          <year>2024</year>
          <fpage>240109637</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2401.09637"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref71">
        <label>71</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wallace</surname>
              <given-names>BC</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Pergola</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Napss: paragraph-level medical text simplification via narrative prompting and sentence-matching summarization</article-title>
          <source>ArXiv. Preprint posted online on February 11, 2023</source>
          <year>2023</year>
          <fpage>230205574</fpage>
          <pub-id pub-id-type="doi">10.18653/v1/2023.findings-eacl.80</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref72">
        <label>72</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Speier</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Ong</surname>
              <given-names>MK</given-names>
            </name>
            <name name-style="western">
              <surname>Arnold</surname>
              <given-names>CW</given-names>
            </name>
          </person-group>
          <article-title>Using phrases and document metadata to improve topic modeling of clinical reports</article-title>
          <source>J Biomed Inform</source>
          <year>2016</year>
          <month>06</month>
          <volume>61</volume>
          <fpage>260</fpage>
          <lpage>6</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(16)30028-4"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2016.04.005</pub-id>
          <pub-id pub-id-type="medline">27109931</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(16)30028-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC4902330</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref73">
        <label>73</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wen</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Nair</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Deng</surname>
              <given-names>CY</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>XH</given-names>
            </name>
            <name name-style="western">
              <surname>Moseley</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>George</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Lindvall</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Mining heterogeneous clinical notes by multi-modal latent topic model</article-title>
          <source>PLoS One</source>
          <year>2021</year>
          <volume>16</volume>
          <issue>4</issue>
          <fpage>e0249622</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pone.0249622"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pone.0249622</pub-id>
          <pub-id pub-id-type="medline">33831055</pub-id>
          <pub-id pub-id-type="pii">PONE-D-20-09700</pub-id>
          <pub-id pub-id-type="pmcid">PMC8031429</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref74">
        <label>74</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zack</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Williams</surname>
              <given-names>CYK</given-names>
            </name>
            <name name-style="western">
              <surname>Sushil</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Butte</surname>
              <given-names>AJ</given-names>
            </name>
          </person-group>
          <article-title>Topic modeling on clinical social work notes for exploring social determinants of health factors</article-title>
          <source>JAMIA Open</source>
          <year>2024</year>
          <month>04</month>
          <volume>7</volume>
          <issue>1</issue>
          <fpage>ooad112</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38223407"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamiaopen/ooad112</pub-id>
          <pub-id pub-id-type="medline">38223407</pub-id>
          <pub-id pub-id-type="pii">ooad112</pub-id>
          <pub-id pub-id-type="pmcid">PMC10788143</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref75">
        <label>75</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Jagannatha</surname>
              <given-names>AN</given-names>
            </name>
            <name name-style="western">
              <surname>Fodeh</surname>
              <given-names>SJ</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Ranking medical terms to support expansion of lay language resources for patient comprehension of electronic health record notes: adapted distant supervision approach</article-title>
          <source>JMIR Med Inform</source>
          <year>2017</year>
          <month>10</month>
          <day>31</day>
          <volume>5</volume>
          <issue>4</issue>
          <fpage>e42</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://medinform.jmir.org/2017/4/e42/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/medinform.8531</pub-id>
          <pub-id pub-id-type="medline">29089288</pub-id>
          <pub-id pub-id-type="pii">v5i4e42</pub-id>
          <pub-id pub-id-type="pmcid">PMC5686421</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref76">
        <label>76</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Kantu</surname>
              <given-names>NS</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Tran</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Duan</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Kwon</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>README: Bridging medical jargon and lay understanding for patient education through data-centric NLP</article-title>
          <source>Find ACL EMNLP</source>
          <year>2024</year>
          <month>11</month>
          <volume>2024</volume>
          <fpage>12609</fpage>
          <lpage>12629</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/2024.findings-emnlp.737</pub-id>
          <pub-id pub-id-type="medline">41710550</pub-id>
          <pub-id pub-id-type="pmcid">PMC12912280</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref77">
        <label>77</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cai</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Bajracharya</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sills</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kapoor</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Berlowitz</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Generation of patient after-visit summaries to support physicians</article-title>
          <year>2022</year>
          <conf-name>Proceedings of the 29th International Conference on Computational Linguistics (COLING)</conf-name>
          <conf-date>October 12-17, 2022</conf-date>
          <conf-loc>Gyeongju, Republic of Korea</conf-loc>
          <publisher-name>International Committee on Computational Linguistics</publisher-name>
          <fpage>6234</fpage>
          <lpage>6247</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://aclanthology.org/2022.coling-1.544"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref78">
        <label>78</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>AQ</given-names>
            </name>
            <name name-style="western">
              <surname>Sablayrolles</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mensch</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Bamford</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Chaplot</surname>
              <given-names>DS</given-names>
            </name>
            <name name-style="western">
              <surname>Casas de las</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Bressand</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>Mistral 7B</article-title>
          <source>ArXiv. Preprint posted online on October 10, 2023</source>
          <year>2023</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2310.06825"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2310.06825</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref79">
        <label>79</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Labrak</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Bazoge</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Morin</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Gourraud</surname>
              <given-names>PA</given-names>
            </name>
            <name name-style="western">
              <surname>Rouvier</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Dufour</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>BioMistral: a collection of open-source pretrained large language models for medical domains</article-title>
          <source>ArXiv. Preprint posted online on February 15, 2024</source>
          <year>2024</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2402.10373"/>
          </comment>
          <pub-id pub-id-type="doi">10.18653/v1/2024.findings-acl.348</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref80">
        <label>80</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grattafiori</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dubey</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Jauhri</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pandey</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kadian</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Al-Dahle</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Letman</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>The Llama 3 herd of models</article-title>
          <source>ArXiv. Preprint posted online on July 31, 2024</source>
          <year>2024</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2407.21783"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref81">
        <label>81</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning</article-title>
          <source>Nature</source>
          <year>2025</year>
          <month>09</month>
          <volume>645</volume>
          <issue>8081</issue>
          <fpage>633</fpage>
          <lpage>638</lpage>
          <pub-id pub-id-type="doi">10.1038/s41586-025-09422-z</pub-id>
          <pub-id pub-id-type="medline">40962978</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-025-09422-z</pub-id>
          <pub-id pub-id-type="pmcid">PMC12443585</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref82">
        <label>82</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>EJ</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wallis</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Allen-Zhu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>LoRA: low-rank adaptation of large language models</article-title>
          <source>ArXiv. Preprint posted online on June 17, 2021</source>
          <year>2021</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2106.09685"/>
          </comment>
          <pub-id pub-id-type="doi">10.5260/chara.21.2.8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref83">
        <label>83</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Li</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Two directions for clinical data generation with large language models: data-to-label and label-to-data</article-title>
          <source>Proc Conf Empir Methods Nat Lang Process</source>
          <year>2023</year>
          <volume>2023</volume>
          <fpage>7129</fpage>
          <lpage>7143</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38213944"/>
          </comment>
          <pub-id pub-id-type="doi">10.18653/v1/2023.findings-emnlp.474</pub-id>
          <pub-id pub-id-type="medline">38213944</pub-id>
          <pub-id pub-id-type="pmcid">PMC10782150</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref84">
        <label>84</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Brown</surname>
              <given-names>TB</given-names>
            </name>
            <name name-style="western">
              <surname>Mann</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Ryder</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Subbiah</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kaplan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Dhariwal</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Language models are few-shot learners</article-title>
          <year>2020</year>
          <conf-name>Proceedings of the 34th International Conference on Neural Information Processing Systems</conf-name>
          <conf-date>December 6-12, 2020</conf-date>
          <conf-loc>Vancouver, BC</conf-loc>
          <fpage>200514165</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://papers.nips.cc/paper_files/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref85">
        <label>85</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>AEW</given-names>
            </name>
            <name name-style="western">
              <surname>Bulgarelli</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Gayles</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Shammout</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Horng</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Pollard</surname>
              <given-names>TJ</given-names>
            </name>
          </person-group>
          <article-title>MIMIC-IV, a freely accessible electronic health record dataset</article-title>
          <source>Sci Data</source>
          <year>2023</year>
          <month>01</month>
          <day>03</day>
          <volume>10</volume>
          <issue>1</issue>
          <fpage>1</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41597-022-01899-x"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41597-022-01899-x</pub-id>
          <pub-id pub-id-type="medline">36596836</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41597-022-01899-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC9810617</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref86">
        <label>86</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sounack</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Davis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Durieux</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Chaffin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pollard</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Lehman</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>AEW</given-names>
            </name>
          </person-group>
          <article-title>BioClinical ModernBERT: a state-of-the-art long-context encoder for biomedical and clinical NLP</article-title>
          <source>ArXiv. Preprint posted online on June 12, 2025</source>
          <year>2025</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.2506.10896</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref87">
        <label>87</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Turney</surname>
              <given-names>PD</given-names>
            </name>
          </person-group>
          <article-title>Learning algorithms for keyphrase extraction</article-title>
          <source>Information Retrieval</source>
          <year>2000</year>
          <month>05</month>
          <volume>2</volume>
          <issue>4</issue>
          <fpage>303</fpage>
          <lpage>336</lpage>
          <pub-id pub-id-type="doi">10.1023/a:1009976227802</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref88">
        <label>88</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zesch</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Gurevych</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Angelova</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Mitkov</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Approximate matching for evaluating keyphrase extraction</article-title>
          <year>2009</year>
          <conf-name>Proceedings of the International Conference on Recent Advances in Natural Language Processing (RANLP-2009)</conf-name>
          <conf-date>June 18, 2024</conf-date>
          <conf-loc>Borovets, Bulgaria</conf-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>484</fpage>
          <lpage>489</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://aclanthology.org/R09-1086.pdf"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref89">
        <label>89</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wolf</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Debut</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Sanh</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Chaumond</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Delangue</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Moi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cistac</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>HuggingFace's transformers: state-of-the-art natural language processing</article-title>
          <source>ArXiv. Preprint posted online on October 9, 2019</source>
          <year>2019</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/1910.03771"/>
          </comment>
          <pub-id pub-id-type="doi">10.18653/v1/2020.emnlp-demos.6</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref90">
        <label>90</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chamieh</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Zesch</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Giebermann</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>LLMs in short answer scoring: limitations and promise of zero-shot and few-shot approaches</article-title>
          <year>2024</year>
          <conf-name>Proceedings of the 19th Workshop on Innovative Use of NLP for Building Educational Applications (BEA 2024)</conf-name>
          <conf-date>June 20, 2024</conf-date>
          <conf-loc>Mexico City, Mexico</conf-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>309</fpage>
          <lpage>315</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://aclanthology.org/2024.bea-1.25/"/>
          </comment>
          <pub-id pub-id-type="doi">10.18653/v1/2025.bea-1.0</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref91">
        <label>91</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Dolan</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Carin</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>What makes good in-context examples for GPT-3?</article-title>
          <source>ArXiv. Preprint posted online on January 17, 2021</source>
          <year>2021</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.2101.06804</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref92">
        <label>92</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Issaiy</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ghanaati</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Kolahi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Shakiba</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Jalali</surname>
              <given-names>AH</given-names>
            </name>
            <name name-style="western">
              <surname>Zarei</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Kazemian</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Methodological insights into ChatGPT's screening performance in systematic reviews</article-title>
          <source>BMC Med Res Methodol</source>
          <year>2024</year>
          <month>03</month>
          <day>27</day>
          <volume>24</volume>
          <issue>1</issue>
          <fpage>78</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedresmethodol.biomedcentral.com/articles/10.1186/s12874-024-02203-8"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12874-024-02203-8</pub-id>
          <pub-id pub-id-type="medline">38539117</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12874-024-02203-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC10976661</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref93">
        <label>93</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Krag</surname>
              <given-names>CH</given-names>
            </name>
            <name name-style="western">
              <surname>Balschmidt</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Bruun</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Brejnebøl</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>JJ</given-names>
            </name>
            <name name-style="western">
              <surname>Boesen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Anderson</surname>
              <given-names>MB</given-names>
            </name>
            <name name-style="western">
              <surname>Muller</surname>
              <given-names>FC</given-names>
            </name>
          </person-group>
          <article-title>Large language models for abstract screening in systematic- and scoping reviews: a diagnostic test accuracy study</article-title>
          <source>medRxiv. Preprint posted online on October 2, 2024</source>
          <year>2024</year>
          <pub-id pub-id-type="doi">10.1101/2024.10.01.24314702</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref94">
        <label>94</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Gupta</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Deng</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>YJ</given-names>
            </name>
            <name name-style="western">
              <surname>Paget</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Naugler</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Automated paper screening for clinical reviews using large language models: data analysis study</article-title>
          <source>J Med Internet Res</source>
          <year>2024</year>
          <month>01</month>
          <day>12</day>
          <volume>26</volume>
          <fpage>e48996</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2024//e48996/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/48996</pub-id>
          <pub-id pub-id-type="medline">38214966</pub-id>
          <pub-id pub-id-type="pii">v26i1e48996</pub-id>
          <pub-id pub-id-type="pmcid">PMC10818236</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref95">
        <label>95</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Homiar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Thomas</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ostinelli</surname>
              <given-names>EG</given-names>
            </name>
            <name name-style="western">
              <surname>Kennett</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Friedrich</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Cuijpers</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Harrer</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Development and evaluation of prompts for a large language model to screen titles and abstracts in a living systematic review</article-title>
          <source>BMJ Ment Health</source>
          <year>2025</year>
          <month>07</month>
          <day>22</day>
          <volume>28</volume>
          <issue>1</issue>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://boris-portal.unibe.ch/handle/20.500.12422/213634"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmjment-2025-301762</pub-id>
          <pub-id pub-id-type="medline">40701625</pub-id>
          <pub-id pub-id-type="pii">bmjment-2025-301762</pub-id>
          <pub-id pub-id-type="pmcid">PMC12306261</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref96">
        <label>96</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vieira</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Allred</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Lankford</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Castilho</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Way</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>How much data is enough data? Fine-tuning large language models for in-house translation: performance evaluation across multiple dataset sizes</article-title>
          <source>ArXiv. Preprint posted online on September 5, 2024</source>
          <year>2024</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.2409.03454</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref97">
        <label>97</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Bosma</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>VY</given-names>
            </name>
            <name name-style="western">
              <surname>Guu</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>AW</given-names>
            </name>
            <name name-style="western">
              <surname>Lester</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Du</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>Finetuned language models are zero-shot learners</article-title>
          <source>ArXiv. Preprint posted online on September 3, 2021</source>
          <year>2022</year>
          <fpage>1</fpage>
          <lpage>42</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://arxiv.org/abs/2109.01652"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2109.01652</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref98">
        <label>98</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jaccard</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>The distribution of the flora in the Alpine zone</article-title>
          <source>New Phytologist</source>
          <year>2006</year>
          <month>05</month>
          <day>05</day>
          <volume>11</volume>
          <issue>2</issue>
          <fpage>37</fpage>
          <lpage>50</lpage>
          <pub-id pub-id-type="doi">10.1111/j.1469-8137.1912.tb05611.x</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref99">
        <label>99</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Han</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>Does synthetic data generation of LLMs help clinical text mining?</article-title>
          <source>ArXiv. Preprint posted on March 8, 2023</source>
          <year>2023</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2303.04360"/>
          </comment>
          <pub-id pub-id-type="doi">10.5260/chara.21.2.8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref100">
        <label>100</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>EDA: easy data augmentation techniques for boosting performance on text classification tasks</article-title>
          <source>ArXiv. Prepint posted online on August 25, 2019</source>
          <year>2019</year>
          <pub-id pub-id-type="doi">10.18653/v1/d19-1670</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref101">
        <label>101</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tam</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Raffel</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Bansal</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>An empirical survey of data augmentation for limited data learning in NLP</article-title>
          <source>ArXiv. Preprint posted online on June 14, 2021</source>
          <year>2021</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2106.07499"/>
          </comment>
          <pub-id pub-id-type="doi">10.1162/tacl_a_00542</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref102">
        <label>102</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Iskander</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cohen</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Karnin</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Shapira</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Tolmach</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Quality matters: evaluating synthetic data for tool-using LLMs</article-title>
          <year>2024</year>
          <conf-name>Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing Miami</conf-name>
          <conf-date>May 19, 2025</conf-date>
          <conf-loc>Florida, USA</conf-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>4958</fpage>
          <lpage>4976</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.285</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref103">
        <label>103</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yuan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Torr</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Real-fake: effective training data synthesis through distribution matching</article-title>
          <source>ArXiv. Preprint posted online on October 16, 2023</source>
          <year>2024</year>
          <fpage>1</fpage>
          <lpage>21</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2310.10402"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2310.10402</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref104">
        <label>104</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Suk</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yue</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Viswanathan</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Gashteovski</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Evaluating language models as synthetic data generators</article-title>
          <source>ArXiv. Preprint posted online on September 1, 2025</source>
          <year>2024</year>
          <pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.320</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref105">
        <label>105</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Hyper-DREAM, a multimodal digital transformation hypertension management platform integrating large language model and digital phenotyping: multicenter development and initial validation study</article-title>
          <source>J Med Syst</source>
          <year>2025</year>
          <month>04</month>
          <day>02</day>
          <volume>49</volume>
          <issue>1</issue>
          <fpage>42</fpage>
          <pub-id pub-id-type="doi">10.1007/s10916-025-02176-1</pub-id>
          <pub-id pub-id-type="medline">40172683</pub-id>
          <pub-id pub-id-type="pii">10.1007/s10916-025-02176-1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref106">
        <label>106</label>
        <nlm-citation citation-type="web">
          <article-title>Enhancing for identifying and prioritizing important medical jargons from electronic health record notes utilizing data augmentation : a pilot study</article-title>
          <source>GitHub</source>
          <access-date>2026-06-23</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://github.com/memy85/2024_medicalnote_annotation">https://github.com/memy85/2024_medicalnote_annotation</ext-link>
          </comment>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
