<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e92716</article-id><article-id pub-id-type="doi">10.2196/92716</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Extracting Quality-of-Life Information of Patients Diagnosed With Breast Cancer From Health Care Online Forum Posts Using Open-Source Large Language Models: Algorithm Development and Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Czok</surname><given-names>Karolina Hanna</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Schmidt</surname><given-names>David Maria</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Brian Po-Han</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kuk</surname><given-names>Deborah</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Feldman</surname><given-names>Josh</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cimiano</surname><given-names>Philipp</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff5">5</xref></contrib></contrib-group><aff id="aff1"><institution>Center for Cognitive Interaction Technology, Faculty of Technology, Bielefeld University</institution><addr-line>Inspiration 1</addr-line><addr-line>Bielefeld</addr-line><country>Germany</country></aff><aff id="aff2"><institution>Inspire</institution><addr-line>Arlington</addr-line><addr-line>VA</addr-line><country>United States</country></aff><aff id="aff3"><institution>Teva Branded Pharmaceutical Products R&#x0026;D LLC</institution><addr-line>Parsippany</addr-line><addr-line>NJ</addr-line><country>United States</country></aff><aff id="aff4"><institution>Datavant</institution><addr-line>Phoenix</addr-line><addr-line>AZ</addr-line><country>United States</country></aff><aff id="aff5"><institution>Semalytix GmbH</institution><addr-line>Bielefeld</addr-line><country>Germany</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Arora</surname><given-names>Akshay</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Maynard</surname><given-names>Diana</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Fluck</surname><given-names>Juliane</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Karolina Hanna Czok, MSc, Center for Cognitive Interaction Technology, Faculty of Technology, Bielefeld University, Inspiration 1, Bielefeld, 33619, Germany, 49 521106 ext 2951; <email>karolina.czok@uni-bielefeld.de</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>5</day><month>10</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e92716</elocation-id><history><date date-type="received"><day>02</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>30</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>30</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Karolina Hanna Czok, David Maria Schmidt, Brian Po-Han Chen, Deborah Kuk, Josh Feldman, Philipp Cimiano. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 5.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e92716"/><abstract><sec><title>Background</title><p>Quality-of-life (QoL) questionnaires are an established instrument designed to assess overall well-being and QoL of patients. They are important in predicting the outcome of the disease and understanding the needs of individual patients. However, their repeated collection imposes a substantial burden on both patients and clinical professionals. Many patients seek emotional support and mutual exchange in online communities for peer support, where they frequently share detailed descriptions of symptoms and treatment experiences, addressing topics covered in QoL questionnaires. The emergence of large language models (LLMs) uncovers potential for automatic extraction of relevant QoL information from patient-generated text.</p></sec><sec><title>Objective</title><p>The aim of this study is to evaluate and compare various open-source LLMs and optimization approaches for automated extraction of QoL information from forum posts.</p></sec><sec sec-type="methods"><title>Methods</title><p>The dataset consisted of 840 English-language posts from patients with breast cancer recruited on Inspire online communities, manually annotated with sentence-level text spans indicating whether and where posts contained information relevant to 53 QoL questions from standardized questionnaires. Eleven open-source LLMs were evaluated in a zero-shot setup under 2 input conditions: post-only and post with additional context. For the GPT-OSS-20B model, additional experiments assessed the impact of chain-of-thought prompting, instruction optimization, few-shot prompting, simultaneous all-questions prompting, and parameter-efficient fine-tuning. For correctly classified yes and no instances, the overlap between model-generated evidence and human-annotated spans was evaluated.</p></sec><sec sec-type="results"><title>Results</title><p>Across 11 evaluated LLMs, Qwen3-14B achieved the highest macro <italic>F</italic><sub>1</sub>-score (0.59) in the zero-shot post-only setting. Providing additional context consistently reduced the performance of all models. Model size did not correlate with <italic>F</italic><sub>1</sub>-score, with several midsized models (14B-30B) outperforming 70B models. For GPT-OSS-20B, chain-of-thought prompting, instruction optimization, and simultaneous all-questions prompting decreased performance. Bootstrap few-shot prompting with random search achieved slightly better performance than the baseline. Parameter-efficient fine-tuning with low-rank adaptation (LoRA) achieved the highest overall performance (0.71). Across all experiments, a strong class imbalance was observed, with models performing substantially better on the majority class &#x201C;not in the text&#x201D; than on the minority classes &#x201C;yes&#x201D; and &#x201C;no.&#x201D; Most classification errors occurred in semantically broad or ambiguous terms and the fallback question. For correctly predicted yes and no answers, model-generated evidence matched or partially matched human-annotated spans in 89% (42/47) of cases.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Automated extraction of QoL information from patient-generated text using open-source LLMs remains a challenging task. While prompt optimization techniques failed to improve baseline zero-shot performance, parameter-efficient fine-tuning with LoRA significantly increased accuracy. However, current models still struggle to reliably detect explicit symptom expressions in heavily imbalanced data, too often predicting the majority class &#x201C;not in the text.&#x201D; Before such tools can be successfully integrated into clinical practice, future research must prioritize strategies to capture these minority-class signals.</p></sec></abstract><kwd-group><kwd>quality of life</kwd><kwd>patient-reported outcomes</kwd><kwd>breast cancer</kwd><kwd>online health care forums</kwd><kwd>social media analysis</kwd><kwd>large language models</kwd><kwd>AI</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Quality of life (QoL), as defined by the World Health Organization, encompasses physical, mental, and social well-being beyond the mere absence of disease [<xref ref-type="bibr" rid="ref1">1</xref>]. Embedding QoL assessments into clinical practice can improve patient satisfaction, increase treatment engagement, help clinicians anticipate risk, and intervene earlier to address key impairments that matter to patients and influence prognosis [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. Traditionally, the gold standard to assess QoL is fixed questionnaires, such as the European Organisation for Research and Treatment of Cancer (EORTC) Quality of Life Questionnaire-Core (QLQ-C30) for patients with cancer [<xref ref-type="bibr" rid="ref4">4</xref>]. Despite the widely recognized value of QoL assessments, their repeated collection imposes a substantial burden on both patients and health care providers [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref8">8</xref>] and therefore remains underused in routine clinical care [<xref ref-type="bibr" rid="ref3">3</xref>].</p><p>Medical information is available on the internet in many different forms, with the internet becoming the second most important source of information after physician consultations [<xref ref-type="bibr" rid="ref9">9</xref>]. Patients facing a chronic or particularly serious illness have an increased need for information and discussion to exchange experiences with other people affected by the same disease [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. Many associations offer forums dedicated to specific types of illnesses, particularly breast cancer, as places for expression, support, and a source of information, allowing users to belong to a community facing the same difficulties [<xref ref-type="bibr" rid="ref12">12</xref>]. These discussions can contain useful information about a patient&#x2019;s well-being, such as QoL information, as shown by Schmidt et al [<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>Breast cancer is the most commonly diagnosed cancer in women and the second most diagnosed cancer overall, with over 2.3 million new cases in 2022 [<xref ref-type="bibr" rid="ref14">14</xref>]. However, approximately 92% of women diagnosed with breast cancer survive at least 5 years postdiagnosis [<xref ref-type="bibr" rid="ref15">15</xref>]. Given the large and growing population of breast cancer survivors, research aimed at understanding their long-term needs is highly relevant [<xref ref-type="bibr" rid="ref16">16</xref>]. The integration of QoL assessments into clinical practice has been shown to provide meaningful benefits for patients with breast cancer, offering them insights into their own care [<xref ref-type="bibr" rid="ref17">17</xref>]. For clinicians, such information can support more individualized treatment decisions by highlighting which interventions may be most beneficial for specific patients based on their circumstances [<xref ref-type="bibr" rid="ref18">18</xref>]. Lim et al [<xref ref-type="bibr" rid="ref19">19</xref>] showed that certain QoL measures, such as better physical functioning, lower pain levels, and reduced appetite loss, are associated with improved survival outcomes in patients with cancer, including those with breast cancer. These findings highlight the value of considering self-reported measures alongside traditional clinical factors in treatment decisions, particularly for patients with breast cancer. Furthermore, women living with and surviving breast cancer report increasing use of a variety of social media platforms as part of their daily routine to manage ongoing care and psychosocial needs [<xref ref-type="bibr" rid="ref20">20</xref>]. Consequently, a substantial amount of potentially valuable patient-generated data exists; however, it remains largely underused in clinical and research settings.</p><p>Recent advancements in AI, particularly the emergence of large language models (LLMs), offer promising avenues to overcome the data collection bottleneck by extracting QoL information from free-text narratives and mapping it to standardized questionnaire items. LLMs have demonstrated promising performance across a range of clinical natural language processing (NLP) tasks involving unstructured text, such as information extraction [<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref23">23</xref>] and summarization [<xref ref-type="bibr" rid="ref24">24</xref>]. The use of locally hosted, open-source LLMs enables compliance with data privacy requirements that are critical for the deployment of health care applications.</p><p>Despite these advances, the application of LLMs to extract QoL information from patient-generated text remains underexplored and, to the authors&#x2019; best knowledge at the time of writing, no prior studies have directly investigated this problem.</p><p>The objective of this study is to address the following research questions (RQs):</p><list list-type="bullet"><list-item><p>RQ1: Among open-source LLMs, which model performs best in a zero-shot setting for extracting QoL questionnaire answers from health forum posts?</p></list-item><list-item><p>RQ2: How do different prompting and optimization strategies (few-shot prompting, instruction optimization, chain-of-thought (CoT) reasoning, simultaneous all-questions prompting, parameter-efficient fine-tuning with low-rank adaptation [LoRA]) influence performance?</p></list-item><list-item><p>RQ3: To what extent can LLMs generate textual evidence that aligns with human-annotated spans supporting yes and no predictions?</p></list-item><list-item><p>RQ4: Which QoL questionnaire items are most prone to misclassification and what types of errors occur most frequently?</p></list-item></list><p>By answering these questions, this study aims to evaluate the potential and limitations of using open-source LLMs on patient-generated forum data as a low-burden and low-cost complement to traditional QoL assessment.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>The study evaluated the feasibility of extracting QoL information from forum posts using LLMs through four phases: (1) model comparison: comparing 11 open-source LLMs of varying sizes (8B to 70B) in post-only and post+context settings; (2) optimization evaluation: assessing CoT prompting, instruction optimization, few-shot prompting, simultaneous all-questions prompting, and parameter-efficient fine-tuning with LoRA for the chosen midsized model; (3) evidence generation: prompting the previously chosen model to provide textual evidence supporting every correct yes and no prediction and comparing this evidence with human-annotated spans; (4) error analysis: calculating error counts per question and per error type. The goal of this framework is to provide a comprehensive assessment of LLMs&#x2019; capabilities in extracting QoL information and to identify potential limitations.</p></sec><sec id="s2-2"><title>Dataset</title><sec id="s2-2-1"><title>Overview</title><p>This study uses the dataset introduced by Schmidt et al [<xref ref-type="bibr" rid="ref13">13</xref>], which comprises three main parts: (1) posts from the Inspire online communities written by users who completed QoL questionnaires; (2) sentence-level annotations of the posts indicating where specific QoL questions are addressed; (3) gold-standard responses to QoL questionnaires, specifically the EORTC QLQ-C30 and 23-item European Organization for Research and Treatment of Cancer Quality of Life Questionnaire - Breast Cancer Module (EORTC QLQ-BR23), provided by the participating users. The EORTC QLQ-C30 is a 30-item questionnaire assessing general QoL in patients with cancer [<xref ref-type="bibr" rid="ref4">4</xref>], while the QLQ-BR23 consists of 23 questions focused on breast cancer&#x2013;specific aspects [<xref ref-type="bibr" rid="ref25">25</xref>]. It is a supplementary module to be used in conjunction with the EORTC QLQ-C30. Inspire is a large online health community consisting of patients and caregivers across a wide range of medical conditions, including cancers. The platform is designed to facilitate open discussion of sensitive health-related topics and to support peer-to-peer interaction among individuals with similar health journeys. Posts are unstructured and vary from short replies to long-form narratives, typically describing symptoms, emotional states, treatment experiences, side effects, and everyday life challenges. Importantly, these posts are independently authored and not written in response to the QoL questionnaires. Instead, they reflect naturally occurring patient experiences.</p><p>The dataset contains 20,204 English-language forum posts and comments from 2006 to 2024, with 580 posts and comments originating within the 6 months preceding the end of the study period. In total, 2683 posts and comments were manually annotated at the sentence level. Among these, 613 contain at least 1 QoL-related annotation. During annotation, trained annotators identified sentences containing information relevant to individual QoL questionnaire items and linked them to the corresponding questions. Annotation was performed by 3 annotators for the most recent 6-month period and 2 annotators for the preceding 18-month period, following the annotation protocol defined in Schmidt et al [<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>The authors have demonstrated that these annotated posts contain substantial information relevant to QoL assessment, showing that user-level questionnaire responses could be predicted from the coded data with an <italic>F</italic><sub>1</sub>-score of approximately 0.70. The 5 most frequently answered questions in the coded data were: &#x201C;Did you feel ill or unwell?,&#x201D; &#x201C;Did you worry?,&#x201D; &#x201C;Have you had pain?,&#x201D; &#x201C;Did you feel tense?&#x201D; and &#x201C;Were you limited in doing either your work or other daily activities?&#x201D; The question &#x201C;Did you feel ill or unwell?&#x201D; was used as a fallback label for symptoms that do not fit into any other question. In the original study, the relationship between forum content and questionnaire responses was analyzed by comparing annotations derived from posts within different temporal windows preceding questionnaire completion, including the most recent 6 months and the most recent 24 months. The authors found that QoL-relevant information contained in forum posts shows substantial agreement with questionnaire responses across both time windows, indicating that patient-generated content can provide meaningful information about QoL over extended periods [<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>In this paper, the EORTC QLQ-C30 and EORTC QLQ-BR23 questionnaires are used exclusively as a source of QoL questions, and the sentence-level annotations are treated as ground truth indicators of whether and where a given post contains information relevant to a particular question. The task is performed at the post level and does not involve predicting full questionnaire outcomes for users based on their complete posting history. Instead, it focuses on determining whether individual QoL questions can be answered based on a single post using LLMs. The task is framed as a 3-way post-level classification. For each of the 53 questionnaire items, a model must assign 1 of 3 distinct target labels: yes (the post confirms the presence of the symptom or experience), no (the post explicitly negates it), or not in the text (the item is completely unaddressed). This formulation enables fine-grained analysis of where specific QoL information is expressed and allows verification of whether the predicted information is correct, as well as characterization of different types of errors made by LLMs. In addition, it allows assessment of whether LLMs can reproduce human-annotated evidence spans for this task.</p><p>For model training and evaluation, the dataset was preprocessed by excluding posts used in a pilot annotation process, posts that described the conditions of individuals other than the posters themselves, such as relatives or friends, and posts specifically referring to distant past events. While posts describing experiences of a third person often contained health-related information, they were not directly relevant to the QoL of the person posting and were thus excluded from the final dataset. The remaining data were split using random sampling into 80% (672/840 posts) train, 10% (84/840 posts) validation, and 10% (84/840 posts) test sets, with the test set used for model evaluation and error analysis. Due to the multilabel nature of the dataset and the long-tail distribution of QoL annotations per post, exact stratification of all label combinations across train, validation, and test splits was not feasible. Many label combinations occur only a few times, which prevents reliable proportional allocation without overfitting the split design. To ensure comparability across splits, we instead verified that the marginal distributions of labels are consistent. In particular, the proportion of posts containing at least 1 QoL annotation is similar across train, validation, and test sets, and the overall label frequency distribution follows the same long-tailed pattern in all splits. Approximately half of the posts in each split contain no QoL annotations (train: 336/672, 50%; validation: 42/84, 50%; and test: 43/84, 51%).</p></sec><sec id="s2-2-2"><title>Corpus Statistics and Class Imbalance</title><p>To provide a granular view of the dataset&#x2019;s composition and highlight the inherent difficulty of the task, we analyze the global distribution of the target labels (&#x201C;yes,&#x201D; &#x201C;no,&#x201D; and &#x201C;not in the text&#x201D;) across the test set. As shown in <xref ref-type="table" rid="table1">Table 1</xref>, the resulting dataset exhibits an extreme class imbalance, characterized by massive sparsity of &#x201C;yes&#x201D; and &#x201C;no&#x201D; labels.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Global class distribution across the test set.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Target label</td><td align="left" valign="bottom">Total instances and percentage of corpus, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Yes</td><td align="left" valign="top">59 (1.33)</td></tr><tr><td align="left" valign="top">No</td><td align="left" valign="top">9 (0.20)</td></tr><tr><td align="left" valign="top">Not in the text</td><td align="left" valign="top">4384 (98.47)</td></tr></tbody></table></table-wrap><p>This highly skewed distribution reflects a real-world challenge in clinical text mining from patient-generated health data: while a patient&#x2019;s overall posting history across months may contain rich QoL insights, any individual forum post or comment typically focuses only on a narrow subset of immediate concerns or symptoms. Consequently, the overwhelming majority of standardized questionnaire items remain unaddressed (not in the text) within a single post. Additionally, patients are naturally more inclined to report the presence of a symptom or experience rather than explicitly state its absence, particularly when responding to a main thread, which accounts for the higher frequency of &#x201C;yes&#x201D; labels relative to &#x201C;no&#x201D; labels.</p></sec><sec id="s2-2-3"><title>Dataset Schema</title><p>The dataset follows a hierarchical structure linking posts, sentence-level evidence spans, and QoL questionnaire items. Each data instance consists of a forum post (or comment) paired with zero or more annotations that map text spans to specific QoL questions.</p><p>At the lowest level, each record contains:</p><list list-type="bullet"><list-item><p>main_post: the main post or comment that the user is replying to (may be null).</p></list-item><list-item><p>content: the original forum post text written by a user textual content associated with the same thread or reply context.</p></list-item><list-item><p>labels: a list of annotated QoL mappings for the post.</p></list-item></list><p>Each element in labels links a span of text to a specific questionnaire item and contains:</p><list list-type="bullet"><list-item><p>question: the QoL questionnaire item (from EORTC QLQ-C30 or EORTC QLQ-BR23).</p></list-item><list-item><p>text: the exact extracted span from the post that provides evidence for the question.</p></list-item><list-item><p>begin, end: character offsets of the annotated span within the post.</p></list-item><list-item><p>exactmatch: whether the span exactly answers the question.</p></list-item><list-item><p>negative: whether the span indicates a negative statement.</p></list-item></list><p>These schema attributes map directly to the target classification labels. If a question contains no associated annotation span for a post, its target label is &#x201C;not in the text.&#x201D; If an annotation span exists and the negative attribute is false, the target label is &#x201C;yes.&#x201D; If an annotation span exists and the negative attribute is true, the target label is &#x201C;no.&#x201D; A single post may contain multiple labels if it includes evidence for multiple QoL aspects or no labels if no QoL question is answered by the post.</p></sec><sec id="s2-2-4"><title>Participants</title><p>The focus of the dataset is patients with breast cancer who have been recruited from Inspire, specifically from the &#x201C;Breast Cancer&#x201D; and &#x201C;Advanced Breast Cancer&#x201D; communities. The inclusion criteria were (1) females with a breast cancer diagnosis, (2) age 18 years or older, (3) residency in the United States, and (4) at least 1 post or comment in the respective Inspire communities.</p><p>In total, 134 participants participated in the study, of which 11 did not have at least 1 post, resulting in 123 participants meeting the inclusion criteria. The average age of the participants was 68.2 (SD 11.5) years, with an average age of 55.7 (SD 12.6) years at the time of the diagnosis. Moreover, at the time of the survey, the average number of years that passed since the diagnosis of the patients was 13.1 (SD 9.4).</p></sec></sec><sec id="s2-3"><title>Selection of LLMs and Experimental Environment</title><p>One of the foremost ethical concerns in AI health care is the security and confidentiality of sensitive health data [<xref ref-type="bibr" rid="ref26">26</xref>]. While commercial LLMs such as GPT (OpenAI) or Gemini (Google) offer powerful linguistic capabilities to support various medical tasks, their cloud-based nature raises significant privacy challenges [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. Using locally hosted, open-source models can provide a privacy-preserving alternative, adhering to health care regulations and ensuring that sensitive medical information remains within the local environment [<xref ref-type="bibr" rid="ref29">29</xref>]. In the first part of the study, 11 open-source base (ie, noninstruct) LLMs representing diverse model families were selected: GPT-OSS (20B) [<xref ref-type="bibr" rid="ref30">30</xref>], Qwen3 (14B, 30B) [<xref ref-type="bibr" rid="ref31">31</xref>], Gemma3 (12B, 27B) [<xref ref-type="bibr" rid="ref32">32</xref>], DeepSeek-R1 (14B, 32B, 70B) [<xref ref-type="bibr" rid="ref33">33</xref>], Llama 3.1 (8B, 70B) [<xref ref-type="bibr" rid="ref34">34</xref>], and Phi-4 (14B) [<xref ref-type="bibr" rid="ref35">35</xref>]. The model sizes ranged from 8B to 70B parameters. Models were run locally via the Ollama [<xref ref-type="bibr" rid="ref36">36</xref>] framework, avoiding transmission of data to cloud-based APIs. Experimental pipelines were implemented using the Declarative Self-improving Python (DSPy) [<xref ref-type="bibr" rid="ref37">37</xref>] declarative framework, ensuring consistent input and output behavior (<xref ref-type="other" rid="box1">Textbox 1</xref>).</p><boxed-text id="box1"><title> Expected input or output behavior specified as a Declarative Self-improving Python signature.</title><p><bold>Input</bold></p><list list-type="bullet"><list-item><p>&#x201C;Content&#x201D;: post or comment of a health care forum user to be investigated.</p></list-item><list-item><p>&#x201C;Main post&#x201D;: context of the health care forum post, for example, the main post in a thread.</p></list-item></list><p>&#x201C;Question&#x201D; : question from the quality-of-life (QoL) questionnaire.</p><p><bold>Output</bold></p><list list-type="bullet"><list-item><p>&#x201C;Answer&#x201D;: answer to the question, either:</p></list-item></list><list list-type="bullet"><list-item><p>&#x201C;not in the text&#x201D;: the information relevant to the question is not mentioned.</p></list-item><list-item><p>&#x201C;no&#x201D;: the post explicitly negates the content of the question.</p></list-item><list-item><p>&#x201C;yes&#x201D;: the post explicitly confirms the content of the question.</p></list-item></list></boxed-text></sec><sec id="s2-4"><title>Zero-Shot Comparison of Models and Input Conditions</title><sec id="s2-4-1"><title>Overview</title><p>The first experiment examined baseline model performance in a zero-shot setting with and without adding context. Adding context means that, in each prompt, the investigated post is accompanied by the main post it refers to within a given thread, that is, the post to which the user is responding. For example, in the post+context setting, the prompt could consist of the reply &#x201C;Yes, I experience it too&#x201D; together with the main post &#x201C;Does any of you have trouble with skin peeling after radiation?&#x201D; In the postonly setting, only the reply &#x201C;Yes, I experience it too&#x201D; would be included in the prompt.</p><p>Each model was evaluated on the test set consisting of 84 posts, paired with the same set of 53 QoL questions from EORTC QLQ-C30 and EORTC QLQ-BR23 questionnaires, generating 4452 post-question predictions per model. For each post, the model was required to predict whether it contains information that would answer the question positively, negatively, or whether the information is not mentioned. An example of the prediction task is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Example of patient-generated &#x201C;content&#x201D; with additional context &#x201C;main post&#x201D; evaluated on 53 &#x201C;questions&#x201D; from European Organisation for Research and Treatment of Cancer (EORTC) Quality of Life Questionnaire-Core 30 (QLQ-C30) and EORTC Quality of Life Questionnaire-Breast Cancer 23 (QLQ-BR23) questionnaires answered with &#x201C;yes, no, not in the text.&#x201D;</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e92716_fig01.png"/></fig><p>To check whether adding the main post from the corresponding discussion thread improves or degrades performance, each model was evaluated twice: once with and once without inclusion of &#x201C;Main post&#x201D; as additional context.</p></sec><sec id="s2-4-2"><title>Evaluation</title><p>For every prediction, the model-generated label was compared against the human-annotated ground truth. However, the extreme class imbalance in this dataset creates a significant evaluation challenge. A na&#x00EF;ve baseline model that always predicts &#x201C;not in the text&#x201D; would automatically achieve an accuracy of 98.47%. Evaluating models through aggregate metrics like weighted <italic>F</italic><sub>1</sub>-score would provide a distorted view of their real performance. To ensure a transparent assessment, we evaluate all models using an unweighted macro <italic>F</italic><sub>1</sub>-score (calculated as the arithmetic mean of the <italic>F</italic><sub>1</sub>-scores for the 3 individual classes) alongside per-class breakdowns.</p><p>For each class (&#x201C;yes,&#x201D; &#x201C;no,&#x201D; &#x201C;not in the text&#x201D;), the <italic>F</italic><sub>1</sub>-score is computed as the harmonic mean of precision and recall [<xref ref-type="bibr" rid="ref38">38</xref>]:</p><disp-formula id="equWL1"><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mtext>=</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x00D7;</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x00D7;</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mtext>/</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mo>+</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>The macro <italic>F</italic><sub>1</sub>-score is then calculated as the arithmetic mean of the 3 classes:</p><disp-formula id="equWL2"><mml:math id="eqn2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="italic">M</mml:mi><mml:mi mathvariant="italic">a</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">r</mml:mi><mml:mi mathvariant="italic">o</mml:mi></mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:msub><mml:mi mathvariant="italic">F</mml:mi><mml:mrow><mml:mn mathvariant="italic">1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>y</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>n</mml:mi><mml:mi>o</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>This ensured that each class contributed equally to the final score, preventing the most frequent &#x201C;not in the text&#x201D; class from dominating the performance assessment and masking minority class failures.</p></sec></sec><sec id="s2-5"><title>Prompt Optimization and Fine-Tuning Experiments</title><sec id="s2-5-1"><title>Experiments Overview</title><p>Following the zero-shot evaluation, one of the midsized models was selected for all subsequent optimization experiments in this section. As this was a feasibility study, restricting experiments to a single model enabled controlled comparisons between optimization methods while minimizing confounding effects related to architectural differences and computational cost.</p><p>A series of optimization experiments were then conducted to assess whether different prompting strategies and lightweight fine-tuning approaches could improve performance in extracting QoL information. All experiments were run under the same task definition and input-output specification as the zero-shot baseline, without providing additional context (Main post). The only variations between experiments were the prompting strategies or additional training applied, ensuring that differences in performance reflected the effect of the optimization technique itself rather than changes in input structure or task formulation. For methodological consistency, all optimization methods were evaluated on the same test split of the dataset using the same macro <italic>F</italic><sub>1</sub>-score as in the baseline experiment.</p></sec><sec id="s2-5-2"><title>CoT Prompting</title><p>CoT is a prompting strategy that mimics the step-by-step thinking ability of humans. It is widely used to decompose multistep problems into intermediate steps, achieving improvement on many reasoning benchmarks by significantly improving the ability of LLMs to perform complex reasoning [<xref ref-type="bibr" rid="ref39">39</xref>]. The experiment consisted of running the DSPy pipeline with the ChainOfThought predictor module, which instructed the model to generate a series of reasoning steps before producing the final label. The reasoning content was not evaluated and only the final classification contributed to the score.</p></sec><sec id="s2-5-3"><title>Instruction Optimization With Multiprompt Instruction PRoposal Optimizer Version 2</title><p>Multiprompt Instruction PRoposal Optimizer Version 2 (MIPROv2) [<xref ref-type="bibr" rid="ref40">40</xref>] is a prompt optimizer capable of optimizing both instructions and few-shot examples jointly in 3 steps:</p><list list-type="bullet"><list-item><p>Bootstrapping few-shot examples, by randomly sampling from the training set and running them through the program. If the output is correct, the example is kept as a candidate. Otherwise, the optimizer tries another example until the specified number of few-shot example candidates is curated.</p></list-item><list-item><p>Generating instruction candidates using a composite prompt that included (1) a summary of the training dataset properties, (2) a summary of the program code and target predictor, (3) previously bootstrapped few-shot examples, and (4) a randomly sampled tip for generation (eg, &#x201C;be creative&#x201D; or &#x201C;be concise.&#x201D;).</p></list-item><list-item><p>Finding the best combination of few-shot examples and instructions using Bayesian optimization. Sets of prompts are evaluated over a validation set for a specified number of trials, with the best set being returned at the end.</p></list-item></list><p>The experiment was performed with 2 variants:</p><list list-type="bullet"><list-item><p xml:lang="en-gb">Zero-shot MIPROv2, which optimizes the instruction program without providing any demonstrations.</p></list-item><list-item><p xml:lang="en-gb">Few-shot MIPROv2, which optimizes the instruction by also including bootstrapped examples.</p></list-item></list><p>Both variants were configured using the medium optimization mode provided by DSPy&#x2019;s MIPROv2 framework. This setting was selected to evaluate optimization behavior under a standard configuration, rather than to exhaustively search the optimization space.</p></sec><sec id="s2-5-4"><title>Few-Shot Prompting</title><p>To evaluate whether a more advanced bootstrapping method would improve the effect of adding few-shot examples to the unoptimized instruction, the next experiment applied DSPy&#x2019;s bootstrap few-shot prompting with random search. This optimization method generates a set of candidate few-shot prompts by combining demonstrations directly from the training set with additional bootstrapped examples produced by a teacher model. It then performs a randomized search over multiple candidate prompt configurations and selects the best-performing program [<xref ref-type="bibr" rid="ref41">41</xref>].</p><p>The configuration was set to include up to 5 labeled and 5 bootstrapped demonstrations per prompt and to evaluate 3 candidate programs. This setting was chosen to limit computational cost while allowing the optimizer to explore multiple prompt configurations within the scope of this feasibility study.</p></sec><sec id="s2-5-5"><title>Fine-Tuning With LoRA</title><p>Fine-tuning is a process of adapting a pretrained model for specialized tasks or domains, by training it on a custom dataset [<xref ref-type="bibr" rid="ref42">42</xref>]. In the context of the medical domain, it enables the model to access specialized clinical terminology, often absent in general language used for pretraining [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>].</p><p>LoRA presents a parameter-efficient approach to fine-tuning LLMs, which enables targeted training without the need to modify the entire model [<xref ref-type="bibr" rid="ref45">45</xref>]. While most of the parameters are fixed, only a small number of parameters are trained to adapt to a new task. This approach requires significantly fewer computational resources than full fine-tuning and avoids altering the base model, thus preserving the pretrained knowledge [<xref ref-type="bibr" rid="ref46">46</xref>].</p><p>Training data consisted of all labeled training examples from the QoL classification dataset. To reduce class imbalance, the minority classes (&#x201C;yes&#x201D; and &#x201C;no&#x201D;) were balanced through random oversampling to match the number of &#x201C;not in the text&#x201D; examples in the training set. During fine-tuning, the model was trained to predict only the target answer, while the prompt tokens were masked from the loss computation. Parameter-efficient fine-tuning was performed for 3 epochs using LoRA with a learning rate of 2&#x00D7;10<sup>&#x2212;5</sup>, an effective batch size of 8, and brain floating point with 16 bits (bfloat16) precision. Only the LoRA adapter parameters were updated, while all base model weights remained fixed. After training, the adapter was saved and used to evaluate the model on the test set under the same conditions as all other experiments.</p></sec><sec id="s2-5-6"><title>Simultaneous All-Questions Prompting</title><p>To investigate the impact of query structure on model performance, simultaneous all-questions prompting was evaluated. In the standard setup, models were queried sequentially, evaluating a single post against a single question at a time. In this alternative configuration, all 53 QoL questions were presented simultaneously within a single prompt. This experiment allows us to determine whether concurrent information extraction increases the models&#x2019; performance compared to single-question prompting.</p></sec></sec><sec id="s2-6"><title>Generation of Textual Evidence and Comparison With Human Annotations</title><p>Due to the complexity of medical data, producing a comprehensive and accurate set of text spans corresponding to specific medical entities often requires involvement of multiple experts with specialized knowledge. The process is therefore costly and resource-intensive, which raises the question of whether LLMs could be used to streamline the process of annotating medical data [<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>].</p><p>The following experiment assessed whether LLMs are capable of generating textual evidence supporting their predictions, a task that could be later extended to producing annotations for the QoL information extraction dataset. For all correctly classified &#x201C;yes&#x201D; and &#x201C;no&#x201D; predictions in the best optimization setting, the chosen model was prompted to provide the exact sentence from the post that justified the answer. The evidence was compared against human-annotated spans contained in the original dataset and categorized into:</p><list list-type="order"><list-item><p>Exact matches - the model-generated evidence span is identical to the corresponding human-annotated text span.</p></list-item><list-item><p>Partial matches - the model-generated evidence was fully contained within the human-annotated text span, or vice versa, and the token overlap was at least 50%.</p></list-item><list-item><p>No matches - the model-generated evidence did not meet the partial match criterion, either because the token overlap was below 50% or the evidence and annotation were not contained within the same sentence.</p></list-item></list><p><xref ref-type="table" rid="table2">Table 2</xref> illustrates the three match categories using examples of model-generated evidence and the corresponding human-annotated spans.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Example of generated evidence for correct &#x201C;yes&#x201D; predictions compared to human-annotated spans.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question</td><td align="left" valign="bottom">True answer</td><td align="left" valign="bottom">Predicted answer</td><td align="left" valign="bottom">Human annotation</td><td align="left" valign="bottom">LLM<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> evidence</td><td align="left" valign="bottom">Type of match</td></tr></thead><tbody><tr><td align="left" valign="top">Have you felt nauseated?</td><td align="left" valign="top">yes</td><td align="left" valign="top">yes</td><td align="left" valign="top">The nausea is much worse during my second chemo cycle.</td><td align="left" valign="top">The nausea is much worse during my second chemo cycle.</td><td align="left" valign="top">Exact match</td></tr><tr><td align="left" valign="top">Were you tired?</td><td align="left" valign="top">yes</td><td align="left" valign="top">yes</td><td align="left" valign="top">Lately I&#x2019;ve been feeling exhausted most days</td><td align="left" valign="top">Lately I&#x2019;ve been feeling exhausted most days, especially in the afternoons</td><td align="left" valign="top">Partial match</td></tr><tr><td align="left" valign="top">Did you feel ill or unwell?</td><td align="left" valign="top">yes</td><td align="left" valign="top">yes</td><td align="left" valign="top">I stopped taking tamoxifen because of the side effects</td><td align="left" valign="top">Some days I just feel unwell and overwhelmed by everything</td><td align="left" valign="top">No match</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap><p>Token overlap was computed as the proportion of tokens shared between the model-generated evidence and the human-annotated span relative to the longer of the 2 spans.</p><p>The proportion of exact and partial evidence matches, relative to the total number of evaluated evidence, was calculated to assess how closely model-generated evidence corresponded to human-identified spans of relevant information.</p></sec><sec id="s2-7"><title>Error Analysis and Question-Level Mispredictions</title><p>The final experiment examined error distribution across QoL questions and the types of errors occurring:</p><list list-type="bullet"><list-item><p>False positives: predicting yes or no when no relevant information was present.</p></list-item><list-item><p>False negatives: predicting not-in-text when relevant information is present.</p></list-item><list-item><p>Label confusion: inverting yes &#x2194; no.</p></list-item></list><p>The predictions for QoL questions with the highest misclassification rates were manually analyzed to identify systematic weaknesses, such as ambiguity or semantic overlap (eg, fatigue vs need to rest).</p></sec><sec id="s2-8"><title>Ethical Considerations</title><p>The study has been approved by the Ethics Committee of Bielefeld University under application number 2023&#x2010;216-W1. Informed consent was obtained from all participants to analyze their posts and comments for this study. The first 100 respondents received a US $20 gift card. The collected data is stored in encrypted form and available only to the authors of the study. All models used in the study are open-source and run in a local environment; therefore, no data have been transferred to the cloud servers.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Baseline Model Comparison in the Zero-Shot Setting</title><p><xref ref-type="table" rid="table3">Table 3</xref> summarizes the macro <italic>F</italic><sub>1</sub>-scores across all 11 evaluated models. Qwen3-14B achieved the highest overall performance in both experimental configurations, reaching a macro <italic>F</italic><sub>1</sub>-score of 0.59 in the post-only setting and 0.56 when context was included. Removing context consistently improved performance of nearly all models, with &#x0394;<italic>F</italic><sub>1</sub> (denoting the increase in macro <italic>F</italic><sub>1</sub>-score after removing context) varying between 0.03 and 0.11. The results do not point to a single model family as universally superior for this task; however, Qwen3 and GPT-OSS families performed slightly better than the alternatives.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Comparison of macro <italic>F</italic><sub>1</sub>-scores with 95% CIs across 11 open-source LLMs<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> for zero-shot QoL<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> information extraction with and without post context with per-class breakdown in the brackets.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Params</td><td align="left" valign="bottom" colspan="4">Post only</td><td align="left" valign="bottom" colspan="4">Postcontext</td><td align="left" valign="bottom">&#x0394;F<sub>1</sub><sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Macro F<sub>1</sub>-score (95% CI)</td><td align="left" valign="top" colspan="3">Class breakdown</td><td align="left" valign="top">Macro F<sub>1</sub>-score (95% CI)</td><td align="left" valign="top" colspan="3">Class breakdown</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Yes</td><td align="left" valign="top">No</td><td align="left" valign="top">Not in the text</td><td align="left" valign="top"/><td align="left" valign="top">Yes</td><td align="left" valign="top">No</td><td align="left" valign="top">Not in the text</td><td align="left" valign="top"/></tr></thead><tbody><tr><td align="left" valign="top">deepseek-r1</td><td align="left" valign="top">70B</td><td align="left" valign="top">0.51 (0.46&#x2010;0.55)</td><td align="left" valign="top">0.42</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.98</td><td align="left" valign="top">0.46 (0.32&#x2010;0.50)</td><td align="left" valign="top">0.34</td><td align="left" valign="top">0.08</td><td align="left" valign="top">0.97</td><td align="left" valign="top">0.05</td></tr><tr><td align="left" valign="top">llama3.1</td><td align="left" valign="top">70B</td><td align="left" valign="top">0.48 (0.44&#x2010;0.53)</td><td align="left" valign="top">0.35</td><td align="left" valign="top">0.13</td><td align="left" valign="top">0.98</td><td align="left" valign="top">0.43 (0.40&#x2010;0.45)</td><td align="left" valign="top">0.25</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.96</td><td align="left" valign="top">0.05</td></tr><tr><td align="left" valign="top">deepseek-r1</td><td align="left" valign="top">32B</td><td align="left" valign="top">0.51 (0.47&#x2010;0.54)</td><td align="left" valign="top">0.47</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.97</td><td align="left" valign="top">0.43 (0.40&#x2010;0.45)</td><td align="left" valign="top">0.32</td><td align="left" valign="top">0.03</td><td align="left" valign="top">0.93</td><td align="left" valign="top">0.08</td></tr><tr><td align="left" valign="top">qwen3</td><td align="left" valign="top">30B</td><td align="left" valign="top">0.52 (0.48&#x2010;0.55)</td><td align="left" valign="top">0.48</td><td align="left" valign="top">0.09</td><td align="left" valign="top">0.98</td><td align="left" valign="top">0.52 (0.48&#x2010;0.57)</td><td align="left" valign="top">0.48</td><td align="left" valign="top">0.11</td><td align="left" valign="top">0.98</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top">gemma3</td><td align="left" valign="top">27B</td><td align="left" valign="top">0.51 (0.44&#x2010;0.58)</td><td align="left" valign="top">0.34</td><td align="left" valign="top">0.22</td><td align="left" valign="top">0.98</td><td align="left" valign="top">0.47 (0.42&#x2010;0.52)</td><td align="left" valign="top">0.27</td><td align="left" valign="top">0.18</td><td align="left" valign="top">0.97</td><td align="left" valign="top">0.04</td></tr><tr><td align="left" valign="top">gpt-oss</td><td align="left" valign="top">20B</td><td align="left" valign="top">0.58 (0.48&#x2010;0.66)</td><td align="left" valign="top">0.42</td><td align="left" valign="top">0.33</td><td align="left" valign="top">0.99</td><td align="left" valign="top">0.54 (0.47&#x2010;0.60)</td><td align="left" valign="top">0.35</td><td align="left" valign="top">0.28</td><td align="left" valign="top">0.98</td><td align="left" valign="top">0.04</td></tr><tr><td align="left" valign="top">qwen3</td><td align="left" valign="top">14B</td><td align="left" valign="top">0.59 (0.51&#x2010;0.66)</td><td align="left" valign="top">0.5</td><td align="left" valign="top">0.29</td><td align="left" valign="top">0.99</td><td align="left" valign="top">0.56 (0.49&#x2010;0.62)</td><td align="left" valign="top">0.46</td><td align="left" valign="top">0.22</td><td align="left" valign="top">0.99</td><td align="left" valign="top">0.03</td></tr><tr><td align="left" valign="top">phi4</td><td align="left" valign="top">14B</td><td align="left" valign="top">0.52 (0.45&#x2010;0.59)</td><td align="left" valign="top">0.42</td><td align="left" valign="top">0.16</td><td align="left" valign="top">0.99</td><td align="left" valign="top">0.41 (0.39&#x2010;0.44)</td><td align="left" valign="top">0.26</td><td align="left" valign="top">0</td><td align="left" valign="top">0.98</td><td align="left" valign="top">0.11</td></tr><tr><td align="left" valign="top">deepseek-r1</td><td align="left" valign="top">14B</td><td align="left" valign="top">0.49 (0.45&#x2010;0.52)</td><td align="left" valign="top">0.42</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.98</td><td align="left" valign="top">0.45 (0.42&#x2010;0.49)</td><td align="left" valign="top">0.31</td><td align="left" valign="top">0.08</td><td align="left" valign="top">0.97</td><td align="left" valign="top">0.04</td></tr><tr><td align="left" valign="top">gemma3</td><td align="left" valign="top">12B</td><td align="left" valign="top">0.45 (0.42&#x2010;0.49)</td><td align="left" valign="top">0.34</td><td align="left" valign="top">0.05</td><td align="left" valign="top">0.98</td><td align="left" valign="top">0.41 (0.39&#x2010;0.44)</td><td align="left" valign="top">0.24</td><td align="left" valign="top">0.03</td><td align="left" valign="top">0.96</td><td align="left" valign="top">0.04</td></tr><tr><td align="left" valign="top">llama3.1</td><td align="left" valign="top">8B</td><td align="left" valign="top">0.36 (0.34&#x2010;0.38)</td><td align="left" valign="top">0.21</td><td align="left" valign="top">0.02</td><td align="left" valign="top">0.85</td><td align="left" valign="top">0.32 (0.31&#x2010;0.33)</td><td align="left" valign="top">0.11</td><td align="left" valign="top">0.01</td><td align="left" valign="top">0.84</td><td align="left" valign="top">0.04</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table3fn2"><p><sup>b</sup>QoL: quality-of-life.</p></fn><fn id="table3fn3"><p><sup>c</sup>&#x0394;<italic>F</italic><sub>1</sub> represents the incremental change in macro <italic>F</italic><sub>1</sub>-score when context is removed, calculated as:<inline-formula><mml:math id="ieqn1"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>F</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:msub><mml:mrow/><mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">y</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>F</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:msub><mml:mrow/><mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi mathvariant="normal">w</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">t</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p></fn></table-wrap-foot></table-wrap><p>A granular analysis of the per-class metrics reveals the underlying difficulty of the task and the reason for the overall poor performance. While all models excelled at predicting the majority &#x201C;not in the text&#x201D; label with nearly all scores exceeding 0.96, performance collapsed on the minority classes. The highest recorded <italic>F</italic><sub>1</sub>-score for the &#x201C;yes&#x201D; class was only 0.50, while performance on the &#x201C;no&#x201D; class was even lower, at just 0.33.</p><p><xref ref-type="fig" rid="figure2">Figure 2</xref> illustrates the relationship between model size and the macro <italic>F</italic><sub>1</sub>-score for 2 different input configurations: post+context (square) and post-only (circle). There is not a simple linear relationship between model size and performance, with several midsized models (14B-30B) outperforming 70B models. Performance generally improves from smaller to midsized models and drops again for larger models.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Macro <italic>F</italic><sub>1</sub>-score as a function of model size by family for 2 input settings. Every model family is represented by a different color. Vertical dashed lines represent the &#x201C;performance penalty&#x201D; from adding context. B: billion.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e92716_fig02.png"/></fig><p>For all subsequent optimization and fine-tuning experiments, GPT-OSS-20B in the post-only input setting was selected as a representative midsized architecture. It demonstrated one of the strongest zero-shot baseline performances among all evaluated models, serving as the baseline from which we would measure the impact of applying optimization strategies.</p></sec><sec id="s3-2"><title>Comparison of Prompt Optimization and Fine-Tuning Strategies</title><p><xref ref-type="table" rid="table4">Table 4</xref> summarizes the performance of GPT-OSS-20B across the different optimization strategies. Contrary to expectations, CoT prompting resulted in lower performance than the plain prediction baseline, decreasing the macro <italic>F</italic><sub>1</sub>-score from 0.58 to 0.52. Similarly, both MIPROv2 optimization strategies failed to outperform the baseline, achieving macro <italic>F</italic><sub>1</sub>-scores of 0.56 in the zero-shot setting and 0.54 in the few-shot setting. Bootstrap few-shot optimization with random search demonstrated the strongest performance among the prompting-based approaches, reaching a macro <italic>F</italic><sub>1</sub>-score of 0.60, representing only a small improvement over the baseline. Prompting all 53 QoL questions simultaneously resulted in the poorest performance (macro <italic>F</italic><sub>1</sub>-score=0.46), indicating that increasing the number of questions within a single prompt negatively affected extraction quality. In contrast, parameter-efficient fine-tuning with LoRA substantially improved predictive performance, achieving the highest overall macro <italic>F</italic><sub>1</sub>-score of 0.71.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Comparison of GPT-OSS-20B performance with 95% CIs under different optimization strategies.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Optimization strategy</td><td align="left" valign="bottom">Macro <italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom" colspan="3">Class breakdown</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Yes</td><td align="left" valign="top">No</td><td align="left" valign="top">Not in the text</td></tr></thead><tbody><tr><td align="left" valign="top">None</td><td align="left" valign="top">0.58 (0.48&#x2010;0.66)</td><td align="left" valign="top">0.42</td><td align="left" valign="top">0.33</td><td align="left" valign="top">0.99</td></tr><tr><td align="left" valign="top">Chain-of-thought prompting</td><td align="left" valign="top">0.52 (0.47&#x2010;0.58)</td><td align="left" valign="top">0.46</td><td align="left" valign="top">0.13</td><td align="left" valign="top">0.99</td></tr><tr><td align="left" valign="top">MIPROv2<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> Zero-shot</td><td align="left" valign="top">0.56 (0.49&#x2010;0.62)</td><td align="left" valign="top">0.42</td><td align="left" valign="top">0.27</td><td align="left" valign="top">0.99</td></tr><tr><td align="left" valign="top">MIPROv2 Few-shot</td><td align="left" valign="top">0.54 (0.47&#x2010;0.62)</td><td align="left" valign="top">0.46</td><td align="left" valign="top">0.21</td><td align="left" valign="top">0.99</td></tr><tr><td align="left" valign="top">Bootstrap few-shot with random search</td><td align="left" valign="top">0.60 (0.50&#x2010;0.69)</td><td align="left" valign="top">0.49</td><td align="left" valign="top">0.32</td><td align="left" valign="top">0.99</td></tr><tr><td align="left" valign="top">All-questions prompting</td><td align="left" valign="top">0.46 (0.42&#x2010;0.50)</td><td align="left" valign="top">0.34</td><td align="left" valign="top">0.05</td><td align="left" valign="top">0.98</td></tr><tr><td align="left" valign="top">Fine-tuning with LoRA<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">0.71 (0.58&#x2010;0.80)</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.47</td><td align="left" valign="top">0.99</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>MIPROv2: Multiprompt Instruction Proposal Optimizer Version 2.</p></fn><fn id="table4fn2"><p><sup>b</sup>LoRA: low-rank adaptation.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Textual Evidence Alignment With Human Annotations</title><p><xref ref-type="table" rid="table5">Table 5</xref> summarizes the alignment of model-generated textual evidence with human annotations. Out of all 4452 predictions, 47 were correctly classified with a &#x201C;yes&#x201D; or &#x201C;no&#x201D; label. In 42 of 47 (89%) predictions, the model-generated evidence either exactly matched or partially matched the human-annotated span. Exact matches were observed in 20 cases (n=47, 42%) and partial matches in 22 cases (n=47, 47%). Only 5 cases (n=47, 11%) were classified as no match for 4 different questions: &#x201C;Did you feel ill or unwell?&#x201D; (2 cases) and &#x201C;Have you had pain?,&#x201D; &#x201C;Have you had any pain in the area of your affected breast?,&#x201D; and &#x201C;Has your physical condition or medical treatment interfered with your social activities?&#x201D; (1 case each).</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Alignment between model-generated and human-annotated evidence spans for correctly classified yes or no predictions.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Match type</td><td align="left" valign="bottom">Values, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Exact match</td><td align="left" valign="top">20 (42)</td></tr><tr><td align="left" valign="top">Partial match</td><td align="left" valign="top">22 (47)</td></tr><tr><td align="left" valign="top">No match</td><td align="left" valign="top">5 (11)</td></tr></tbody></table></table-wrap></sec><sec id="s3-4"><title>Misclassification Patterns Across QoL Questions</title><p>Analysis of model errors in the baseline setting revealed that a primary difficulty occurred when deciding whether relevant information was present in the text. Approximately 76% (85/112) of misclassifications constituted the model predicting a positive or negative answer for questions with the correct label &#x201C;not in the text.&#x201D; The remaining 24% (27/112) of errors occurred when the model predicted &#x201C;not in the text&#x201D; despite the presence of information supporting a &#x201C;yes&#x201D; or &#x201C;no&#x201D; answer. No cases were observed in which the model generated a reversed answer (ie, predicting &#x201C;yes&#x201D; instead of &#x201C;no&#x201D; or vice versa).</p><p>A less prominent pattern was observed for the LoRA fine-tuned model. 40% (20/50) of LoRA errors corresponded to predicting &#x201C;not in the text&#x201D; despite evidence supporting a &#x201C;yes&#x201D; or &#x201C;no&#x201D; answer, while 58% (29/50) corresponded to predicting a positive or negative answer when the correct label was &#x201C;not in the text.&#x201D; Only 2% (1/50) of errors involved inversion of &#x201C;yes&#x201D; and &#x201C;no&#x201D; labels.</p><p>Errors were unevenly distributed across QoL questions, as shown in <xref ref-type="table" rid="table6">Table 6</xref>. The fallback question &#x201C;Did you feel ill or unwell?&#x201D; was the most error-prone question in the dataset, with 22 errors (26% error rate). The second and third most misclassified questions were &#x201C;Did you worry?&#x201D; with 10 errors (12% error rate) and &#x201C;Have you had pain?&#x201D; with 9 errors (11% error rate).</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Top 10 QoL questionnaire items with the highest number of incorrect predictions.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question</td><td align="left" valign="bottom">Error count</td></tr></thead><tbody><tr><td align="left" valign="top">Did you feel ill or unwell?</td><td align="left" valign="top">22</td></tr><tr><td align="left" valign="top">Did you worry?</td><td align="left" valign="top">10</td></tr><tr><td align="left" valign="top">Have you had pain?</td><td align="left" valign="top">9</td></tr><tr><td align="left" valign="top">Were you limited in doing either your work or other daily activities?</td><td align="left" valign="top">6</td></tr><tr><td align="left" valign="top">Did you need to rest?</td><td align="left" valign="top">6</td></tr><tr><td align="left" valign="top">Were you worried about your health in the future?</td><td align="left" valign="top">5</td></tr><tr><td align="left" valign="top">Were you tired?</td><td align="left" valign="top">5</td></tr><tr><td align="left" valign="top">Did pain interfere with your daily activities?</td><td align="left" valign="top">4</td></tr><tr><td align="left" valign="top">Have you felt weak?</td><td align="left" valign="top">4</td></tr><tr><td align="left" valign="top">Did you feel depressed?</td><td align="left" valign="top">3</td></tr></tbody></table></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><p>This study investigated whether open-source LLMs can be used as a supplementary tool for extracting QoL information from online forum posts, using a breast cancer forum as an example, in a way that aligns with standardized questionnaire responses. It further explored which model and optimization technique may be best-suited for this task, assessed whether LLMs can potentially reproduce human annotations, and investigated possible challenges based on the types of errors occurring for specific QoL items.</p><p>Open-source LLMs demonstrated poor ability to identify QoL-relevant information from patient-generated text, achieving macro-averaged <italic>F</italic><sub>1</sub>-scores between 0.32&#x2010;0.56 in a zero-shot setting with added context and 0.36&#x2010;0.59 without added context. This suggests that QoL information extraction remains a challenging task for current open-source LLMs without task-specific training. Overall performance depends on the model family and size; careful choice of model may therefore be a crucial starting point in the methodology design. Models from the GPT-OSS and Qwen families demonstrated higher potential for successfully extracting QoL information than other models. Notably, increasing the model size did not result in the highest scores, suggesting that increased parameter count alone does not guarantee better performance for this task. From a practical perspective, the stronger performance of midsized models compared to larger 70B models may be especially advantageous in clinical practice, as these models typically require fewer computational resources and less specialized hardware for deployment. However, inference time and resource consumption were not evaluated in this study and remain important directions for future work.</p><p>The analysis of per-class <italic>F</italic><sub>1</sub>-scores further reveals a strong imbalance in performance across classes, with models achieving substantially higher performance on the majority class &#x201C;not in the text&#x201D; compared to the minority classes &#x201C;yes&#x201D; and &#x201C;no,&#x201D; which highlights a systematic weakness when using zero-shot LLMs for QoL information extraction from health care forum posts. From a clinical safety perspective, the models&#x2019; strong performance on the &#x201C;not in the text&#x201D; label is beneficial. In health care environments, minimizing false positives prevents the system from hallucinating nonexistent symptoms, which could lead to inappropriate clinical conclusions. However, the primary objective of deploying LLMs in this context is the successful extraction of explicit &#x201C;yes&#x201D; and &#x201C;no&#x201D; signals. In a real-world clinical monitoring workflow, failing to detect instances where a patient reports severe treatment side effects or acute emotional distress introduces critical false negatives. Missing these signals directly compromises the ability to track a patient&#x2019;s QoL, making a zero-shot extraction framework insufficient for practical clinical deployment.</p><p>Providing additional context (the main post to which the investigated post refers) consistently led to decreased performance, suggesting that additional conversational context may introduce noise rather than clarifying information for QoL information extraction. Confusion matrices for the post-only and post+context settings (<xref ref-type="fig" rid="figure3">Figure 3</xref>) show that including context increased the number of misclassifications, particularly by incorrectly assigning &#x201C;yes&#x201D; or &#x201C;no&#x201D; labels to posts containing no information about the symptoms. This observation was supported by a manual analysis of prediction errors, which showed that approximately 37% of errors in the post+context setting resulted from confusion between the target post and the main post. For example, in 1 instance the main post contained the statement &#x201C;I've been on CO2 for 2 years &#x0026; never felt depressed. Suddenly I'm having terrible anxiety attacks. I'm barely eating or sleeping...,&#x201D; while the corresponding target post consisted of another user&#x2019;s recommendation to seek therapy for anxiety. For the question &#x201C;Have you had trouble sleeping?,&#x201D; the correct label was &#x201C;not in the text,&#x201D; as the question referred to the author of the target post rather than the person speaking in the main post. However, when conversational context was provided, the model predicted &#x201C;yes,&#x201D; suggesting that it relied on information from the context instead. Adding context should therefore be avoided for this task or the methodology should be carefully designed to prevent confusion, particularly if future studies suggest that removing context results in loss of relevant information.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Confusion matrices for zero-shot quality of life (QoL) classification in post-only and post+context settings.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e92716_fig03.png"/></fig><p>The evaluation of prompting-based optimization strategies applied to GPT-OSS-20B showed that prompt engineering methods provide only limited improvements over the baseline. CoT prompting resulted in a performance decrease compared to the plain zero-shot baseline, which aligns with findings by Liu et al [<xref ref-type="bibr" rid="ref49">49</xref>] that CoT can reduce performance on tasks where thinking makes humans worse-here reflected as &#x201C;overthinking&#x201D; and assigning QoL questions to indirect symptoms. The qualitative examples in (<xref ref-type="table" rid="table7">Table 7</xref>) provide evidence for this effect, showing cases where the model shifts from direct symptom interpretation to more speculative reasoning paths, which leads to incorrect final predictions. Instruction optimization using MIPROv2 resulted in a slight performance decrease compared to the baseline in both zero-shot and few-shot settings. Bootstrap few-shot prompting with random search achieved the best results among prompt-based methods; however, the improvement over the baseline remained marginal. A possible explanation for the limited effectiveness of prompt-based optimization techniques is that the task requires fine-grained matching between patient-generated texts and specific QoL questions, while prompt optimization methods mainly improve general instruction-following or reasoning behavior. In particular, methods such as MIPROv2 optimize a single global prompt over the entire dataset, which may not be well suited for this setting due to the high variability across the 53 QoL questions, which differ in the symptoms they target and in how explicitly they are expressed in patient language. As a result, a single optimized prompt may not improve performance consistently across all questions and can even introduce more general or abstract reasoning patterns that are not helpful for identifying symptom mentions. While carefully selected few-shot demonstrations can provide some task-specific guidance, they are still not sufficient to cover the diversity of both patient-generated posts and QoL question formulations.</p><table-wrap id="t7" position="float"><label>Table 7.</label><caption><p>Examples of chain-of-tCoT reasoning errors and resulting misclassifications.</p></caption><table id="table7" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Post</td><td align="left" valign="bottom">Question</td><td align="left" valign="bottom">True answer</td><td align="left" valign="bottom">Predicted answer</td><td align="left" valign="bottom">Reasoning</td></tr></thead><tbody><tr><td align="left" valign="top">Everyone&#x2019;s descriptions above of side effects mimic my experiences with fatigue and esp the metallic taste. Nausea meds were pretty ineffective compared to now. That being said, my body reacted about the same to each treatment and I had 9 mos of treatment. I knew that by day 4 after the treatment I would be feeling better and could eat more. Knowing how my body would react helped me handle the effects. Obviously, the drugs worked for me and the cancer had a very low 5-year survival rate, so I have been blessed.</td><td align="left" valign="top">Have you felt weak?</td><td align="left" valign="top">not in the text</td><td align="left" valign="top">yes</td><td align="left" valign="top">The user explicitly mentions experiencing fatigue as a side effect of the chemotherapy. Fatigue is commonly associated with feeling weak. Therefore, the post indicates that the user has felt weak.</td></tr><tr><td align="left" valign="top">I had 5 weeks of radiation that covered the entire breast and under the arm where the sentinel node biopsy was done. I went Monday through Friday, and other than the fatigue and, of course, the terrible redness and discomfort from the treatment, it was actually okay.</td><td align="left" valign="top">Do you need to stay in bed or a chair during the day?</td><td align="left" valign="top">not in the text</td><td align="left" valign="top">no</td><td align="left" valign="top">The post states that the user underwent 5 weeks of radiation, attending sessions Monday through Friday. The user mentions fatigue and discomfort but indicates that overall it was &#x201C;actually okay.&#x201D; This implies that the user was not confined to bed or a chair during the day; they were able to attend treatment sessions regularly. Therefore, the post answers the question by indicating that staying in bed or a chair during the day is not necessary.</td></tr><tr><td align="left" valign="top">Hello all, Does anyone have any ideas regarding low calcium levels in 24-hour urine testing? My Calcium Oxalate Saturation was well below normal, and my Calcium Phosphate and Calcium/Creatinine Ratio were also low. Chloride was lower than normal and PH was higher than normal. I, of course, will ask my endo doc when I see her on October 1st but I have tried to research reasons for these low levels and can not find any answers. I have been diagnosed with osteopenia. Does anyone have any ideas regarding these results?</td><td align="left" valign="top">Were you worried about your health in the future?</td><td align="left" valign="top">not in the text</td><td align="left" valign="top">yes</td><td align="left" valign="top">The user is actively seeking explanations for abnormal lab results and expresses concern about their health by asking for ideas and noting they will consult a doctor. This indicates worry about their future health.</td></tr></tbody></table></table-wrap><p>In contrast, parameter-efficient fine-tuning with LoRA resulted in a substantial improvement in performance compared to all prompt-based methods. This suggests that adapting model parameters is more effective for this task than modifying prompts, likely because the model is able to learn task-specific patterns directly from the training data instead of relying on general instruction templates. In particular, fine-tuning allows the model to better capture subtle differences between similar QoL questions and to adjust to the specific language used in patient-generated text. The improvement is especially visible in the minority class &#x201C;yes&#x201D; which indicates that fine-tuning helps the model better identify explicit symptom mentions that were often missed in the zero-shot setting. Despite the overall improvement, the results also suggest several possible directions for improving the fine-tuning setup. In this study, class imbalance was addressed using simple random oversampling, which effectively duplicates the limited number of &#x201C;yes&#x201D; and &#x201C;no&#x201D; examples. While this helps to balance class frequencies, it does not increase the diversity of training signals and may therefore limit the model&#x2019;s ability to generalize to new expressions of symptoms. More advanced data augmentation strategies, such as synthetic generation of additional training examples, could potentially provide more diverse and informative training data and further improve performance. In addition, in this study, a single LoRA adapter was trained across all 53 QoL questions. While this approach is computationally efficient, it may limit the model&#x2019;s ability to specialize for different types of questions, which vary in symptom type and linguistic formulation. Training separate adapters for different subsets of questions, or even per-question adapters, could potentially improve performance by allowing more targeted adaptation. However, such approaches would significantly increase computational cost and were beyond the scope of this study.</p><p>The analysis of textual evidence demonstrated that the tested model could successfully justify its predictions with supporting text. In nearly 90% of correctly classified yes and no cases, model-generated evidence overlapped with human-annotated spans, either exactly or partially. This finding is particularly relevant for clinical interpretability but also highlights the potential of LLMs to generate synthetic annotations that align with human annotations, which could reduce manual effort and increase the availability of labeled data for optimization methods that benefit from larger training samples.</p><p>Error analysis revealed systematic weaknesses related to mapping free-text symptoms description to discrete questionnaire items. Most misclassifications occurred when the model inferred an answer despite the absence of explicit information in the text, especially for a fallback question &#x201C;Did you feel ill or unwell?&#x201D; which exhibited a notably higher error rate (26%) compared to other questions. This is likely related to the semantic role of this label in the annotation scheme, as it functions as a broad catchall category for cases that do not match more specific symptom questions, as well as for general statements of wellness or illness [<xref ref-type="bibr" rid="ref13">13</xref>]. As such, it often captures semantically underspecified or heterogeneous expressions, which may be inherently more difficult to map to a single EORTC QoL construct. Symptoms such as fever or low blood pressure did not fit into other predefined categories and were therefore annotated with this fallback label. However, for a model performing question-wise prediction, it is not clear whether a more specific matching category exists. Emotional and symptom-related questions with overlapping semantics (eg, pain vs discomfort, fatigue vs need for rest) were particularly prone to error. These findings highlight the challenge of defining clear decision rules for assigning free-text expressions to specific QoL questionnaire items. Even for human annotators, it may be difficult to determine whether indirect expressions (eg, <italic>&#x201C;</italic>I constantly needed to rest&#x201D;) should be interpreted as evidence for the question (&#x201C;Were you tired?&#x201D;) or whether only explicit mentions (eg, &#x201C;I felt very tired&#x201D;) should be considered correct. Overlapping questions, such as general worry and worry about health in the future, further complicate the task, as answering both questions positively may introduce redundancy, while restricting the answer to only one requires nuanced rules that are not defined in the questionnaire structure. These findings are consistent with Schmidt et al [<xref ref-type="bibr" rid="ref13">13</xref>], who reported similar difficulties during the annotation process of the dataset used in this study.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>A foundational work directly related to this research is that of Schmidt et al [<xref ref-type="bibr" rid="ref13">13</xref>], which introduced a feasibility study for extracting QoL information from cancer forum posts from Inspire.com and comparing it to patient-reported outcomes on the EORTC QLQ-C30 and EORTC QLQ-BR23 questionnaires. This study demonstrated that QoL information is present in online forum posts and that manual coding can reliably link free-text content with structured questionnaire responses. However, it did not explore the potential of using AI to automate the process.</p><p>A systematic review by L&#x00E1;zaro et al [<xref ref-type="bibr" rid="ref50">50</xref>] synthesized the scientific evidence on the application of NLP techniques as a QoL analysis tool in patients with chronic conditions, including cancer. However, this review covered studies published between 2011 and 2021 and therefore did not include recent advances in NLP which are part of this work. Individual studies mentioned in the review have addressed related but narrower tasks, such as detecting depressive symptoms in online forums (Karmen et al [<xref ref-type="bibr" rid="ref51">51</xref>]), extracting patient-reported breast cancer symptoms from free-text electronic health record notes (Forsyth et al [<xref ref-type="bibr" rid="ref21">21</xref>]), and detecting and quantifying pain in radiation oncology office notes of patients with cancer with bone metastases (Naseri et al [<xref ref-type="bibr" rid="ref52">52</xref>]). While these studies demonstrate the feasibility of extracting QoL-related information from unstructured text, they focused on a specific subset of symptoms and did not attempt to map them to standardized QoL questionnaires.</p><p>More recently, researchers have begun to evaluate using LLMs for structured information extraction in clinical context, although with different task scopes and data sources than this study. For example, Garcia-Carmona et al [<xref ref-type="bibr" rid="ref22">22</xref>] assessed the performance of 6 LLMs (both proprietary and open-source) in a zero-shot setting in extracting patient demographics, diagnostic details, and pharmacological data from unstructured medical reports. Lee et al [<xref ref-type="bibr" rid="ref23">23</xref>] compared the performance of human reviewers and a locally deployed LLM for extracting key thyroid cancer histologic and staging information from surgical pathology reports. While these studies explore the potential of using LLMs for clinical NLP tasks, they focus on texts authored by medical professionals rather than patient-generated text and do not address QoL information.</p><p>A closely related study by Nair et al [<xref ref-type="bibr" rid="ref24">24</xref>] evaluated multiple LLMs and prompting strategies for summarizing patient-generated content from web-based forums and health communities related to breast cancer. The authors compared zero-shot, few-shot, and CoT prompting results with manual reference summaries. They found that few-shot prompting outperformed other approaches, which aligns with observations from this study. However, summarization tasks differ fundamentally from extracting information to answer standardized questionnaire items.</p><p>In contrast to existing work, this study provides a systematic evaluation of open-source LLMs and optimization strategies for extracting questionnaire-aligned QoL information from patient-generated forum posts, using a breast cancer forum as an example. Unlike prior research focusing on different tasks, clinical text, or narrow symptom extraction, this study evaluates prediction accuracy for all symptoms included in EORTC QLQ-C30 and EORTC QLQ-BR23 questionnaires. Additionally, it investigates the potential of using LLMs for recreating human annotations by extracting text spans supporting the prediction. These contributions support the potential use of open-source LLMs in a real-world clinical setting for low-burden, interpretable QoL information extraction from patient-generated text.</p></sec><sec id="s4-3"><title>Limitations</title><p>This study has several limitations that readers should be aware of when interpreting the results. First, the dataset was derived from a single online breast cancer forum and included a limited number of patients and posts. Additionally, all participants were US residents, as Inspire.com members are primarily US-based. As a result, the findings may not be generalizable to other diseases or overall patient populations, where language use and symptom reporting may differ. Different patient communities may vary in how explicitly or implicitly symptoms are described and in how medical terminology is used, which may affect the mapping between forum posts and EORTC QoL measures. Future studies should assess whether similar performance patterns hold across other diseases, languages, and online platforms. This is especially important for non-English or culturally distinct forums, where symptom expression and linguistic framing may differ substantially and may not align with the assumptions embedded in EORTC-based labeling schemes.</p><p>Second, ground-truth labels were derived from manual annotations that required mapping patient-generated forum content to structured questionnaire items. As discussed earlier, some texts are overlapping or ambiguous, which poses difficulties even for human annotators. The mean Fleiss&#x2019; kappa for the data used in this study was 0.5, indicating moderate to high inter-annotator agreement. This remaining uncertainty may have affected the quality of the dataset and, consequently, both model performance estimates and error analyses.</p><p>Third, the dataset was restricted to posts where the author describes their own condition. Posts referring to other individuals (eg, relatives or friends) were excluded to maintain a clear definition of the target variable (QoL of the post author). While this improves label consistency, it artificially simplifies the task. In real-world applications, systems would need to first identify whether a post refers to the author or to another person before applying QoL classification.</p><p>Furthermore, this study focused exclusively on locally hosted, open-source LLMs. While this reflects realistic constraints in clinical settings due to privacy concerns, it limits direct comparison with proprietary models that may achieve different performance levels. In addition, only a subset of prompting and fine-tuning strategies was explored, and hyperparameters were not exhaustively tuned. All optimization experiments were conducted using a single model (GPT-OSS-20B) for controlled comparison, but the findings may not generalize to other families and sizes.</p><p>Finally, parameter-efficient fine-tuning was conducted using a single LoRA adapter shared across all 53 QoL questions. This design choice, made for computational efficiency, may limit the ability of the model to specialize across question types.</p></sec><sec id="s4-4"><title>Conclusions</title><p>In conclusion, this study demonstrates that extracting QoL information from patient-generated health care forum posts using open-source LLMs is a challenging task with limited and highly variable performance. Without task-specific training, baseline zero-shot models perform poorly. They consistently struggle to detect explicit symptom information (&#x201C;yes&#x201D; and &#x201C;no&#x201D; responses), which is the primary objective of this extraction task, and instead largely default to the majority &#x201C;not in the text&#x201D; class.</p><p>We found that prompt-based optimization strategies provide little to no benefit over zero-shot baselines. Parameter-efficient fine-tuning with LoRA proved significantly more effective, achieving the highest overall accuracy and better adapting to the patient-generated text. However, performance on the minority classes remains limited even after fine-tuning. Because missing an explicit symptom report in a real-world setting means failing to detect potential treatment side effects or emotional distress, automated QoL extraction with open-source LLMs is not feasible for clinical deployment. While fine-tuning offers a promising path forward, future work must focus on ensuring the reliable detection of minority-class signals.</p></sec></sec></body><back><ack><p>The authors thank the European Organisation for Research and Treatment of Cancer for giving us the permission to use their questionnaires QLQ-C30 and BR-23 (request IDs 96009 and 102328). We are grateful to Dr Regina Stodden and Viju Sudhi for reviewing the paper and providing helpful suggestions.</p><p>Authors BC and DK were employed by Inspire during the majority of the study period. BC&#x2019;s employment with Inspire ended in August 2025, and DK&#x2019;s employment with Inspire ended in September 2025.</p></ack><notes><sec><title>Funding</title><p>This work was partially funded by the European Union and the German state North RhineWestphalia within the following project of the European Regional Development Fund (EFRE): LLM4KMU - Optimized use of open source large language models in SMEs. It was also partially funded by the Ministry of Culture and Science of the State of North Rhine-Westphalia under grant NW21-059A (SAIL). The funders had no involvement in the study design, data collection, analysis, interpretation, or the writing of the manuscript.</p></sec><sec><title>Data Availability</title><p>The dataset used in this study is not publicly available due to the sensitive nature of the data and the potential risk of participant re-identification. In accordance with the informed consent process, participants were assured that their data would remain confidential and accessible only to the primary research team. Detailed information regarding the dataset characteristics, the original data collection methodology, annotation guidelines, and insights from annotators are available in the study by Schmidt et al [<xref ref-type="bibr" rid="ref13">13</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: BC, DK, DMS, KHC, PC</p><p>Data curation: BC, DK, DMS, JF</p><p>Formal analysis: KHC</p><p>Funding acquisition: PC</p><p>Investigation: KHC</p><p>Methodology: KHC</p><p>Software: KHC (lead), DMS (supporting)</p><p>Supervision: PC</p><p>Writing &#x2013; original draft: KHC (lead), DMS (supporting)</p><p>Writing &#x2013; review &#x0026; editing: KHC (lead), BC (supporting), DK (supporting), DMS (supporting), JF (supporting), PC (supporting)</p></fn><fn fn-type="conflict"><p>PC is a cofounder and shareholder of Semalytix GmbH, a company offering social media listening services for patient-focused drug development. BC and DK are former employees of Inspire, the online community that served as the source of the dataset used in this study.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">bfloat16</term><def><p>brain floating point with 16 bits</p></def></def-item><def-item><term id="abb2">CoT</term><def><p>Chain-of-thought</p></def></def-item><def-item><term id="abb3">DSPy</term><def><p>Declarative Self-improving Python</p></def></def-item><def-item><term id="abb4">EORCTC QLQ-BR23</term><def><p>23-item European Organization for Research and Treatment of Cancer Quality of Life Questionnaire - Breast Cancer Module</p></def></def-item><def-item><term id="abb5">EORTC</term><def><p>European Organization for Research and Treatment of Cancer</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">LoRA</term><def><p>low-rank adaptation</p></def></def-item><def-item><term id="abb8">MIPROv2</term><def><p>Multiprompt Instruction Proposal Optimizer Version 2</p></def></def-item><def-item><term id="abb9">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb10">QLQ-C30</term><def><p>Quality of Life Questionnaire-Core</p></def></def-item><def-item><term id="abb11">QoL</term><def><p>quality-of-life</p></def></def-item><def-item><term id="abb12">RQ</term><def><p>research question</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>The Constitution of the World Health Organisation</article-title><source>Global Health &#x0026; Human Rights Database</source><year>1948</year><access-date>2026-09-20</access-date><publisher-name>WHO</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.globalhealthrights.org/instrument/constitution-of-the-world-health-organization-who/">https://www.globalhealthrights.org/instrument/constitution-of-the-world-health-organization-who/</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Boyer</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lan&#x00E7;on</surname><given-names>C</given-names> </name><name name-style="western"><surname>Baumstarck</surname><given-names>K</given-names> </name><name name-style="western"><surname>Parola</surname><given-names>N</given-names> </name><name name-style="western"><surname>Berbis</surname><given-names>J</given-names> </name><name name-style="western"><surname>Auquier</surname><given-names>P</given-names> </name></person-group><article-title>Evaluating the impact of a quality of life assessment with feedback to clinicians in patients with schizophrenia: randomised controlled trial</article-title><source>Br J Psychiatry</source><year>2013</year><month>06</month><volume>202</volume><issue>6</issue><fpage>447</fpage><lpage>453</lpage><pub-id pub-id-type="doi">10.1192/bjp.bp.112.123463</pub-id><pub-id pub-id-type="medline">23661768</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fernandes</surname><given-names>S</given-names> </name><name name-style="western"><surname>Boussat</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Harnessing generative AI for quality of life assessment: foundations for a new research agenda</article-title><source>J Epidemiol Popul Health</source><year>2025</year><month>10</month><volume>73</volume><issue>5</issue><fpage>203156</fpage><pub-id pub-id-type="doi">10.1016/j.jeph.2025.203156</pub-id><pub-id pub-id-type="medline">41271395</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aaronson</surname><given-names>NK</given-names> </name><name name-style="western"><surname>Ahmedzai</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bergman</surname><given-names>B</given-names> </name><etal/></person-group><article-title>The European Organization for Research and Treatment of Cancer QLQ-C30: a quality-of-life instrument for use in international clinical trials in oncology</article-title><source>J Natl Cancer Inst</source><year>1993</year><month>03</month><day>3</day><volume>85</volume><issue>5</issue><fpage>365</fpage><lpage>376</lpage><pub-id pub-id-type="doi">10.1093/jnci/85.5.365</pub-id><pub-id pub-id-type="medline">8433390</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Detmar</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Aaronson</surname><given-names>NK</given-names> </name></person-group><article-title>Quality of life assessment in daily clinical oncology practice: a feasibility study</article-title><source>Eur J Cancer</source><year>1998</year><month>07</month><volume>34</volume><issue>8</issue><fpage>1181</fpage><lpage>1186</lpage><pub-id pub-id-type="doi">10.1016/s0959-8049(98)00018-5</pub-id><pub-id pub-id-type="medline">9849476</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bezjak</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>P</given-names> </name><name name-style="western"><surname>Skeel</surname><given-names>R</given-names> </name><name name-style="western"><surname>Depetrillo</surname><given-names>AD</given-names> </name><name name-style="western"><surname>Comis</surname><given-names>R</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>KM</given-names> </name></person-group><article-title>Oncologists&#x2019; use of quality of life information: results of a survey of Eastern Cooperative Oncology Group physicians</article-title><source>Qual Life Res</source><year>2001</year><volume>10</volume><issue>1</issue><fpage>1</fpage><lpage>13</lpage><pub-id pub-id-type="doi">10.1023/a:1016692804023</pub-id><pub-id pub-id-type="medline">11508471</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Buxton</surname><given-names>J</given-names> </name><name name-style="western"><surname>White</surname><given-names>M</given-names> </name><name name-style="western"><surname>Osoba</surname><given-names>D</given-names> </name></person-group><article-title>Patients&#x2019; experiences using a computerized program with a touch-sensitive video monitor for the assessment of health-related quality of life</article-title><source>Qual Life Res</source><year>1998</year><month>08</month><volume>7</volume><issue>6</issue><fpage>513</fpage><lpage>519</lpage><pub-id pub-id-type="doi">10.1023/a:1008826408328</pub-id><pub-id pub-id-type="medline">9737141</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Velikova</surname><given-names>G</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Selby</surname><given-names>PJ</given-names> </name></person-group><article-title>Computer-based quality of life questionnaires may contribute to doctor-patient interactions in oncology</article-title><source>Br J Cancer</source><year>2002</year><month>01</month><day>7</day><volume>86</volume><issue>1</issue><fpage>51</fpage><lpage>59</lpage><pub-id pub-id-type="doi">10.1038/sj.bjc.6600001</pub-id><pub-id pub-id-type="medline">11857011</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Natalia</surname><given-names>P</given-names> </name><name name-style="western"><surname>Alejandro</surname><given-names>V</given-names> </name><name name-style="western"><surname>Konstantina</surname><given-names>K</given-names> </name><name name-style="western"><surname>C&#x00E9;lia</surname><given-names>B</given-names> </name></person-group><article-title>Online health information search: what struggles and empowers the users? results of an online survey</article-title><source>Studies in Health Technology and Informatics</source><year>2012</year><publisher-name>IOS Press</publisher-name><pub-id pub-id-type="doi">10.3233/978-1-61499-101-4-843</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raupach</surname><given-names>JCA</given-names> </name><name name-style="western"><surname>Hiller</surname><given-names>JE</given-names> </name></person-group><article-title>Information and support for women following the primary treatment of breast cancer</article-title><source>Health Expect</source><year>2002</year><month>12</month><volume>5</volume><issue>4</issue><fpage>289</fpage><lpage>301</lpage><pub-id pub-id-type="doi">10.1046/j.1369-6513.2002.00191.x</pub-id><pub-id pub-id-type="medline">12460218</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Batta&#x00EF;a</surname><given-names>C</given-names> </name></person-group><article-title>Information m&#x00E9;dicale et &#x00E9;motion dans les forums de sant&#x00E9;</article-title><source>LCN</source><year>2016</year><month>06</month><day>30</day><volume>12</volume><issue>1-2</issue><fpage>51</fpage><lpage>71</lpage><pub-id pub-id-type="doi">10.3166/lcn.12.1-2.51-71</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burgu&#x00E9;</surname><given-names>H</given-names> </name><name name-style="western"><surname>Trensz</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mathelin</surname><given-names>C</given-names> </name><name name-style="western"><surname>Schohn</surname><given-names>A</given-names> </name></person-group><article-title>Les forums de discussion d&#x00E9;di&#x00E9;s au cancer du sein peuvent-ils &#x00EA;tre utiles aux soignants? analyse des messages initiaux du forum de la Ligue nationale contre le cancer pendant une ann&#x00E9;e</article-title><source>Gynecol Obstet Fertil Senol</source><year>2024</year><month>07</month><volume>52</volume><issue>7-8</issue><fpage>466</fpage><lpage>472</lpage><pub-id pub-id-type="doi">10.1016/j.gofs.2024.02.003</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmidt</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Schubert</surname><given-names>R</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>BPH</given-names> </name><etal/></person-group><article-title>Extracting quality of life information of patients diagnosed with breast cancer from health care online forum posts: data feasibility study</article-title><source>JMIR Cancer</source><year>2026</year><month>04</month><day>30</day><volume>12</volume><issue>1</issue><fpage>e76044</fpage><pub-id pub-id-type="doi">10.2196/76044</pub-id><pub-id pub-id-type="medline">42060536</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bray</surname><given-names>F</given-names> </name><name name-style="western"><surname>Laversanne</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sung</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Global cancer statistics 2022: GLOBOCAN estimates of incidence and mortality worldwide for 36 cancers in 185 countries</article-title><source>CA Cancer J Clin</source><year>2024</year><volume>74</volume><issue>3</issue><fpage>229</fpage><lpage>263</lpage><pub-id pub-id-type="doi">10.3322/caac.21834</pub-id><pub-id pub-id-type="medline">38572751</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><article-title>Survival rates for breast cancer</article-title><source>American Cancer Society</source><access-date>2026-06-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.cancer.org/cancer/types/breast-cancer/understanding-a-breast-cancer-diagnosis/breast-cancer-survival-rates.html">https://www.cancer.org/cancer/types/breast-cancer/understanding-a-breast-cancer-diagnosis/breast-cancer-survival-rates.html</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hamer</surname><given-names>J</given-names> </name><name name-style="western"><surname>McDonald</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Quality of life (QOL) and symptom burden (SB) in patients with breast cancer</article-title><source>Support Care Cancer</source><year>2017</year><month>02</month><volume>25</volume><issue>2</issue><fpage>409</fpage><lpage>419</lpage><pub-id pub-id-type="doi">10.1007/s00520-016-3417-6</pub-id><pub-id pub-id-type="medline">27696078</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Perry</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kowalski</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>CH</given-names> </name></person-group><article-title>Quality of life assessment in women with breast cancer: benefits, acceptability and utilization</article-title><source>Health Qual Life Outcomes</source><year>2007</year><month>05</month><day>2</day><volume>5</volume><issue>1</issue><fpage>24</fpage><pub-id pub-id-type="doi">10.1186/1477-7525-5-24</pub-id><pub-id pub-id-type="medline">17474993</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nolazco</surname><given-names>JI</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>SL</given-names> </name></person-group><article-title>The role of health-related quality of life in improving cancer outcomes</article-title><source>J Clin Transl Res</source><year>2023</year><month>04</month><day>28</day><volume>9</volume><issue>2</issue><fpage>110</fpage><lpage>114</lpage><pub-id pub-id-type="medline">37179791</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lim</surname><given-names>L</given-names> </name><name name-style="western"><surname>Machingura</surname><given-names>A</given-names> </name><name name-style="western"><surname>Taye</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Prognostic value of baseline EORTC QLQ-C30 scores for overall survival across 46 clinical trials covering 17 cancer types: a validation study</article-title><source>EClinicalMedicine</source><year>2025</year><month>04</month><volume>82</volume><fpage>103153</fpage><pub-id pub-id-type="doi">10.1016/j.eclinm.2025.103153</pub-id><pub-id pub-id-type="medline">40201799</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aristokleous</surname><given-names>I</given-names> </name><name name-style="western"><surname>Karakatsanis</surname><given-names>A</given-names> </name><name name-style="western"><surname>Masannat</surname><given-names>YA</given-names> </name><name name-style="western"><surname>Kastora</surname><given-names>SL</given-names> </name></person-group><article-title>The role of social media in breast cancer care and survivorship: a narrative review</article-title><source>Breast Care (Basel)</source><year>2023</year><month>04</month><volume>18</volume><issue>3</issue><fpage>193</fpage><lpage>199</lpage><pub-id pub-id-type="doi">10.1159/000531136</pub-id><pub-id pub-id-type="medline">37404835</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Forsyth</surname><given-names>AW</given-names> </name><name name-style="western"><surname>Barzilay</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hughes</surname><given-names>KS</given-names> </name><etal/></person-group><article-title>Machine learning methods to extract documentation of breast cancer symptoms from electronic health records</article-title><source>J Pain Symptom Manage</source><year>2018</year><month>06</month><volume>55</volume><issue>6</issue><fpage>1492</fpage><lpage>1499</lpage><pub-id pub-id-type="doi">10.1016/j.jpainsymman.2018.02.016</pub-id><pub-id pub-id-type="medline">29496537</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garcia-Carmona</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Prieto</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Puertas</surname><given-names>E</given-names> </name><name name-style="western"><surname>Beunza</surname><given-names>JJ</given-names> </name></person-group><article-title>Leveraging large language models for accurate retrieval of patient information from medical reports: systematic evaluation study</article-title><source>JMIR AI</source><year>2025</year><month>07</month><day>3</day><volume>4</volume><issue>1</issue><fpage>e68776</fpage><pub-id pub-id-type="doi">10.2196/68776</pub-id><pub-id pub-id-type="medline">40608403</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>D</given-names> </name><name name-style="western"><surname>Vaid</surname><given-names>A</given-names> </name><name name-style="western"><surname>Menon</surname><given-names>KM</given-names> </name><etal/></person-group><article-title>Using large language models to automate data extraction from surgical pathology reports: retrospective cohort study</article-title><source>JMIR Form Res</source><year>2025</year><month>04</month><day>7</day><volume>9</volume><issue>1</issue><fpage>e64544</fpage><pub-id pub-id-type="doi">10.2196/64544</pub-id><pub-id pub-id-type="medline">40194317</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nair</surname><given-names>RAS</given-names> </name><name name-style="western"><surname>Hartung</surname><given-names>M</given-names> </name><name name-style="western"><surname>Heinisch</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Summarizing online patient conversations using generative language models: experimental and comparative study</article-title><source>JMIR Med Inform</source><year>2025</year><month>04</month><day>14</day><volume>13</volume><issue>1</issue><fpage>e62909</fpage><pub-id pub-id-type="doi">10.2196/62909</pub-id><pub-id pub-id-type="medline">40228244</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sprangers</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Groenvold</surname><given-names>M</given-names> </name><name name-style="western"><surname>Arraras</surname><given-names>JI</given-names> </name><etal/></person-group><article-title>The European Organization for Research and Treatment of Cancer breast cancer-specific quality-of-life questionnaire module: first results from a three-country field study</article-title><source>J Clin Oncol</source><year>1996</year><month>10</month><volume>14</volume><issue>10</issue><fpage>2756</fpage><lpage>2768</lpage><pub-id pub-id-type="doi">10.1200/JCO.1996.14.10.2756</pub-id><pub-id pub-id-type="medline">8874337</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maleki Varnosfaderani</surname><given-names>S</given-names> </name><name name-style="western"><surname>Forouzanfar</surname><given-names>M</given-names> </name></person-group><article-title>The role of AI in hospitals and clinics: transforming healthcare in the 21st century</article-title><source>Bioengineering (Basel)</source><year>2024</year><month>03</month><day>29</day><volume>11</volume><issue>4</issue><fpage>337</fpage><pub-id pub-id-type="doi">10.3390/bioengineering11040337</pub-id><pub-id pub-id-type="medline">38671759</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mesk&#x00F3;</surname><given-names>B</given-names> </name><name name-style="western"><surname>Topol</surname><given-names>EJ</given-names> </name></person-group><article-title>The imperative for regulatory oversight of large language models (or generative AI) in healthcare</article-title><source>NPJ Digit Med</source><year>2023</year><month>07</month><day>6</day><volume>6</volume><issue>1</issue><fpage>120</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00873-0</pub-id><pub-id pub-id-type="medline">37414860</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSJ</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sandmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hegselmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fujarski</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Benchmark evaluation of DeepSeek large language models in clinical decision-making</article-title><source>Nat Med</source><year>2025</year><month>08</month><volume>31</volume><issue>8</issue><fpage>2546</fpage><lpage>2549</lpage><pub-id pub-id-type="doi">10.1038/s41591-025-03727-2</pub-id><pub-id pub-id-type="medline">40267970</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Agarwal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ahmad</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ai</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Gpt-oss-120b &#x0026; gpt-oss-20b model card</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 8, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2508.10925</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Li</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Qwen3 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  May 14, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.09388</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Team</surname><given-names>G</given-names> </name><name name-style="western"><surname>Kamath</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ferret</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Gemma 3 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 25, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2503.19786</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="other"><person-group person-group-type="author"><collab>DeepSeek-AI</collab><name name-style="western"><surname>Guo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name><etal/></person-group><article-title>DeepSeek-R1: incentivizing reasoning capability in llms via reinforcement learning</article-title><comment>Preprint posted online on  Jan 22, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.12948</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Grattafiori</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dubey</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jauhri</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The llama 3 herd of models</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 31, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2407.21783</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Abdin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Aneja</surname><given-names>J</given-names> </name><name name-style="western"><surname>Behl</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Phi-4 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 12, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.08905</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="web"><article-title>Ollama</article-title><source>Github</source><year>2025</year><access-date>2025-12-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/ollama/ollama">https://github.com/ollama/ollama</ext-link></comment></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Khattab</surname><given-names>O</given-names> </name><name name-style="western"><surname>Singhvi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Maheshwari</surname><given-names>P</given-names> </name><etal/></person-group><article-title>DSPy: compiling declarative language model calls into self-improving pipelines</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 5, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2310.03714</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="book"><source>Advanced Data Mining Techniques</source><year>2008</year><publisher-name>Springer</publisher-name><pub-id pub-id-type="doi">10.1007/978-3-540-76917-0</pub-id><pub-id pub-id-type="other">9783540769163</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Schuurmans</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Chain-of-thought prompting elicits reasoning in large language models</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 10, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2201.11903</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Opsahl-Ong</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ryan</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Purtell</surname><given-names>J</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Chen</surname><given-names>YN</given-names> </name><name name-style="western"><surname>Bansal</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>YN</given-names> </name></person-group><article-title>Optimizing instructions and demonstrations for multi-stage language model programs</article-title><access-date>2026-09-08</access-date><conf-name>Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 12-16, 2024</conf-date><conf-loc>Miami, FL</conf-loc><fpage>9340</fpage><lpage>9366</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2024.emnlp-main">https://aclanthology.org/2024.emnlp-main</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.525</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="web"><article-title>Prompt optimizing with GEPA</article-title><source>DSPy</source><access-date>2025-12-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://dspy.ai/learn/optimization/optimizers/">https://dspy.ai/learn/optimization/optimizers/</ext-link></comment></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Anisuzzaman</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Malins</surname><given-names>JG</given-names> </name><name name-style="western"><surname>Friedman</surname><given-names>PA</given-names> </name><name name-style="western"><surname>Attia</surname><given-names>ZI</given-names> </name></person-group><article-title>Fine-tuning large language models for specialized use cases</article-title><source>Mayo Clin Proc Digit Health</source><year>2025</year><month>03</month><volume>3</volume><issue>1</issue><fpage>100184</fpage><pub-id pub-id-type="doi">10.1016/j.mcpdig.2024.11.005</pub-id><pub-id pub-id-type="medline">40206998</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McIntosh</surname><given-names>TR</given-names> </name><name name-style="western"><surname>Susnjak</surname><given-names>T</given-names> </name><name name-style="western"><surname>Arachchilage</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Inadequacies of large language model benchmarks in the era of generative artificial intelligence</article-title><source>IEEE Trans Artif Intell</source><year>2026</year><month>01</month><volume>7</volume><issue>1</issue><fpage>22</fpage><lpage>39</lpage><pub-id pub-id-type="doi">10.1109/TAI.2025.3569516</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>N</given-names> </name></person-group><article-title>Large language models in health care: Development, applications, and challenges</article-title><source>Health Care Sci</source><year>2023</year><month>08</month><volume>2</volume><issue>4</issue><fpage>255</fpage><lpage>263</lpage><pub-id pub-id-type="doi">10.1002/hcs2.61</pub-id><pub-id pub-id-type="medline">38939520</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wallis</surname><given-names>P</given-names> </name><etal/></person-group><article-title>LoRA: low-rank adaptation of large language models</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 16, 2021</comment><pub-id pub-id-type="doi">10.48550/arXiv.2106.09685</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ponkshe</surname><given-names>K</given-names> </name><name name-style="western"><surname>Vepakomma</surname><given-names>P</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Che</surname><given-names>W</given-names> </name><name name-style="western"><surname>Nabende</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shutova</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pilehvar</surname><given-names>MT</given-names> </name></person-group><article-title>FedEx-lora: exact aggregation for federated and efficient fine-tuning of large language models</article-title><conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><conf-loc>Vienna, Austria</conf-loc><fpage>1316</fpage><lpage>1336</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.67</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Goel</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gueta</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gilon</surname><given-names>O</given-names> </name><etal/></person-group><article-title>LLMs Accelerate annotation for medical information extraction</article-title><source>Arxiv</source><comment>Preprint posted online on  Dec 4, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2312.02296</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Tan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Chen</surname><given-names>YN</given-names> </name><name name-style="western"><surname>Bansal</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>YN</given-names> </name></person-group><article-title>Large language models for data annotation and synthesis: a survey</article-title><conf-name>Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 12-16, 2024</conf-date><conf-loc>Miami, FL</conf-loc><fpage>930</fpage><lpage>957</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.54</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Geng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>AJ</given-names> </name><etal/></person-group><article-title>Mind your step (by step): chain-of-thought can reduce performance on tasks where thinking makes humans worse</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 13, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2410.21333</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>L&#x00E1;zaro</surname><given-names>E</given-names> </name><name name-style="western"><surname>Yepez</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Mar&#x00ED;n-Maicas</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Efficiency of natural language processing as a tool for analysing quality of life in patients with chronic diseases. A systematic review</article-title><source>Comput Hum Behav Rep</source><year>2024</year><month>05</month><volume>14</volume><fpage>100407</fpage><pub-id pub-id-type="doi">10.1016/j.chbr.2024.100407</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Karmen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Hsiung</surname><given-names>RC</given-names> </name><name name-style="western"><surname>Wetter</surname><given-names>T</given-names> </name></person-group><article-title>Screening Internet forum participants for depression symptoms by assembling and enhancing multiple NLP methods</article-title><source>Comput Methods Programs Biomed</source><year>2015</year><month>06</month><volume>120</volume><issue>1</issue><fpage>27</fpage><lpage>36</lpage><pub-id pub-id-type="doi">10.1016/j.cmpb.2015.03.008</pub-id><pub-id pub-id-type="medline">25891366</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naseri</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kafi</surname><given-names>K</given-names> </name><name name-style="western"><surname>Skamene</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Development of a generalizable natural language processing pipeline to extract physician-reported pain from clinical reports: generated using publicly-available datasets and tested on institutional clinical reports for cancer patients with bone metastases</article-title><source>J Biomed Inform</source><year>2021</year><month>08</month><volume>120</volume><fpage>103864</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2021.103864</pub-id><pub-id pub-id-type="medline">34265451</pub-id></nlm-citation></ref></ref-list></back></article>