<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e95081</article-id><article-id pub-id-type="doi">10.2196/95081</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Augmenting Head and Neck Multidisciplinary Tumor Board Recommendations With Locally Run Large Language Models: Prospective Evaluation of Real-World Implementation</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Buhr</surname><given-names>Christoph Raphael</given-names></name><degrees>MBA, MD, Dr</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>M&#x00FC;ller</surname><given-names>Lukas</given-names></name><degrees>MBA, MD, Dr, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pinto dos Santos</surname><given-names>Daniel</given-names></name><degrees>MD, PD Dr</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Thiem</surname><given-names>Daniel</given-names></name><degrees>MHBA, MD, PD Dr Dr</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ponciano</surname><given-names>Jean-Jacques</given-names></name><degrees>MSc, Dr</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kr&#x00FC;ger</surname><given-names>Maximilian</given-names></name><degrees>MD, PD Dr Dr</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>O'Brien</surname><given-names>Karoline</given-names></name><degrees>MD, Dr</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kaufmann</surname><given-names>Justus</given-names></name><degrees>MD, Dr</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Nolte</surname><given-names>Hildegard</given-names></name><degrees>MD, Dr</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gartenschl&#x00E4;eger</surname><given-names>Martin</given-names></name><degrees>MD, Dr</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zimmer</surname><given-names>Stefanie</given-names></name><degrees>MD, Dr</degrees><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Altmann</surname><given-names>Sebastian</given-names></name><degrees>MD, PD Dr</degrees><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Blaikie</surname><given-names>Andrew</given-names></name><degrees>BSC, MBchB, FRCOphth, PhD, MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ruckes</surname><given-names>Christian</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff9">9</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Matthias</surname><given-names>Christoph</given-names></name><degrees>MD, Prof Dr</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kuhn</surname><given-names>Sebastian</given-names></name><degrees>MME, MD, Prof Dr</degrees><xref ref-type="aff" rid="aff10">10</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Eckrich</surname><given-names>Jonas</given-names></name><degrees>MD, Dr</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Otorhinolaryngology, University Medical Center of the Johannes Gutenberg-University Mainz</institution><addr-line>Langenbeckstra&#x00DF;e 1</addr-line><addr-line>Mainz</addr-line><country>Germany</country></aff><aff id="aff2"><institution>University of St Andrews, School of Medicine</institution><addr-line>St Andrews</addr-line><country>United Kingdom</country></aff><aff id="aff3"><institution>Department of Diagnostic and Interventional Radiology, University Medical Center of the Johannes Gutenberg-University Mainz</institution><addr-line>Mainz</addr-line><country>Germany</country></aff><aff id="aff4"><institution>Department of Oral and Maxillofacial Surgery-Plastic Surgery, University Medical Center Mainz of the Johannes Gutenberg-University Mainz</institution><addr-line>Mainz</addr-line><country>Germany</country></aff><aff id="aff5"><institution>Department of Radiotherapy and Oncology, University Medical Center of the Johannes Gutenberg-University Mainz</institution><addr-line>Mainz</addr-line><country>Germany</country></aff><aff id="aff6"><institution>Department of Hematology &#x0026; Medical Oncology, University Medical Center Mainz of the Johannes Gutenberg-University Mainz</institution><addr-line>Mainz</addr-line><country>Germany</country></aff><aff id="aff7"><institution>Institute of Pathology, University Medical Center of the Johannes Gutenberg-University Mainz</institution><addr-line>Mainz</addr-line><country>Germany</country></aff><aff id="aff8"><institution>Department Neuroradiology, University Medical Center Mainz of the Johannes Gutenberg-University Mainz</institution><addr-line>Mainz</addr-line><country>Germany</country></aff><aff id="aff9"><institution>Interdisciplinary Center for Clinical Trials (IZKS), University Medical Center of the Johannes Gutenberg-University Mainz</institution><addr-line>Mainz</addr-line><country>Germany</country></aff><aff id="aff10"><institution>Institute for Digital Medicine Philipps-University Marburg and University Hospital of Giessen and Marburg, Marburg</institution><addr-line>Marburg</addr-line><country>Germany</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Gonz&#x00E1;lez</surname><given-names>Liz</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Cao</surname><given-names>Yuchen</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Su</surname><given-names>Zhaohui</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Christoph Raphael Buhr, MBA, MD, Dr, Department of Otorhinolaryngology, University Medical Center of the Johannes Gutenberg-University Mainz, Langenbeckstra&#x00DF;e 1, Mainz, Germany, 49 6131 17 7362; <email>buhrchri@uni-mainz.de</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>20</day><month>8</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e95081</elocation-id><history><date date-type="received"><day>10</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>29</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>29</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Christoph Raphael Buhr, Lukas M&#x00FC;ller, Daniel Pinto dos Santos, Daniel Thiem, Jean-Jacques Ponciano, Maximilian Kr&#x00FC;ger, Karoline O'Brien, Justus Kaufmann, Hildegard Nolte, Martin Gartenschlaeger, Stefanie Zimmer, Sebastian Altmann, Andrew Blaikie, Christian Ruckes, Christoph Matthias, Sebastian Kuhn, Jonas Eckrich. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 20.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e95081"/><abstract><sec><title>Background</title><p>Multidisciplinary tumor boards (MDTs) constitute the foundation of modern tumor therapy. Large language models (LLMs) are widely discussed for optimizing their recommendations.</p></sec><sec><title>Objective</title><p>This is the first prospective feasibility study evaluating the implementation of locally run LLMs on real-world cases within a regular head and neck MDT.</p></sec><sec sec-type="methods"><title>Methods</title><p>Seventeen patients participated in the study. The MDT cases were processed by 2 different local LLMs (gemma-3-12b and gpt-oss-20b) to obtain treatment recommendations. The MDT conferred as usual. After the decision was made, the MDT was presented with the LLMs&#x2019; recommendations. If deemed to be beneficial, the MDT&#x2019;s recommendation was adjusted. The MDT members rated the LLMs&#x2019; responses inter alia, for medical adequacy on a 6-point Likert scale. In addition, a tabular comparison of the MDT&#x2019;s and LLMs&#x2019; recommendations was carried out.</p></sec><sec sec-type="results"><title>Results</title><p>In one case, 6% (1/17, 95% CI 0%&#x2010;29%), the LLM was able to substantially improve the MDT recommendation by underscoring a follow-up examination that had not yet been performed. Concordance regarding the curative or palliative therapy regimen reached 94% (16/17, 95% CI 71%&#x2010;100%); for gemma-3-12b and 59% (10/17, 95% CI 33%&#x2010;82%) for gpt-oss-20b. Gemma-3-12b stated the same first-line therapy regimen as the MDT as first-line in 35% (6/17, 95% CI 14%&#x2010;62%) of cases, and gpt-oss-20b in 41% (7/17, 95% CI 18%&#x2010;67%) of cases. In 59% (10/17, 95% CI 33%&#x2010;82%) of patients, gemma-3-12b stated the MDT&#x2019;s first-line therapy regimen, albeit with a different priority, while for gpt-oss-20b, it was 41% (7/17, 95% CI 18%&#x2010;67%) of patients. Medical adequacy, as rated by the MDT members, revealed a median of 5 (IQR 2&#x2010;5) for gemma-3-12b and 4 (IQR 3&#x2010;5) for gpt-oss-20b. MDT members stated potentially hazardous information in 27% (25/93, 95% CI 18%&#x2010;37%) of ratings for gemma-3-12b and 17% (14/83, 95% CI 9%&#x2010;26%) of ratings for gpt-oss-20b.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Locally run LLMs improved the MDT recommendation in 1 case and were primarily useful for identifying potentially relevant missing information in other cases, underscoring that they cannot replace MDTs. However, their observed benefit suggests that more advanced local models may offer safe, rapid, and cost-effective support for MDT decision-making. The study should be seen as an exploratory setting focusing on practical insights rather than benchmarking its clinical impact. Accordingly, the data demonstrate that the integration of LLMs in today&#x2019;s MDT workflow is feasible and may benefit the quality of decision-making in specific cases.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>LLM</kwd><kwd>AI</kwd><kwd>otorhinolaryngology</kwd><kwd>ORL</kwd><kwd>head and neck</kwd><kwd>digital health</kwd><kwd>chatbot</kwd><kwd>language model</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Multidisciplinary Tumor Board</title><p>Multidisciplinary tumor board (MDT) recommendations have a central influence on the fate of patients with tumors and form the foundation of therapeutic decision-making in modern tumor therapy. Accordingly, optimizing these treatment decisions is a field of great scientific relevance. In recent years, the use of large language models (LLMs) for MDTs has gained increasing attention.</p></sec><sec id="s1-2"><title>LLM Types and Terminology</title><p>LLM systems relevant to MDT support can be grouped along three practical dimensions: (1) deployment (cloud-based vs on-premises), (2) model accessibility (closed-weight/vendor models vs open-weight models that can be self-hosted), and (3) model size, where small open-weight models (approximately 7-20B parameters) can often run on standard local hardware, whereas large open-weight models (eg, 70B+parameters) typically require server-grade graphics processing units (GPUs). A further safety-relevant distinction is whether outputs are generated from the model alone (ungrounded), merely instructed to follow guidelines (guideline-instructed), or explicitly grounded in the full text of authoritative guidance (eg, via retrieval-augmented generation that provides the relevant guideline passages at inference time).</p></sec><sec id="s1-3"><title>Web-Based LLMs in Head and Neck MDTs</title><p>Different studies have evaluated the performance of web-based LLMs in the context of head and neck MDTs [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. Lechien et al [<xref ref-type="bibr" rid="ref1">1</xref>] assessed ChatGPT-4 on 20 medical records of patients with head and neck cancer regarding additional examinations, management, and therapeutic approaches. The authors report an accurate therapeutic proposition for 65% (13/20) of the cases and a significantly higher number of additional examinations than recommended by practitioners, concluding that LLMs may be an adjunctive theoretical tool in simple oncological board decisions. Schmidl et al [<xref ref-type="bibr" rid="ref2">2</xref>] shared details of 20 cases from an MDT with ChatGPT 3.5 and ChatGPT 4.0, reporting that the LLMs provided significantly more treatment options than the MDT. However, due to incorrect treatment options in some instances, the authors conclude that the LLMs are currently only suitable as supporting tools. A further study by the same workgroup assessed Claude 3 Opus and ChatGPT 4.0 on primary head and neck cancer cases [<xref ref-type="bibr" rid="ref3">3</xref>]. Here, the authors describe a similar performance of the models regarding clinical recommendations, explanation, and summarization, but a superior performance of Claude 3 Opus over ChatGPT 4.0 for diagnostic workup and treatment recommendations. Vural Camalan et al [<xref ref-type="bibr" rid="ref4">4</xref>] compared ChatGPT-o1 and DeepSeek V3 on simulated cases, stating correct treatment recommendations in 62% of cases for ChatGPT-o1 and 80% of cases for DeepSeek V3.</p></sec><sec id="s1-4"><title>Addressing Data Protection Aspects by Locally Run LLMs</title><p>Data protection regulations, including the Health Insurance Portability and Accountability Act in the United States [<xref ref-type="bibr" rid="ref5">5</xref>] and the General Data Protection Regulation in Europe [<xref ref-type="bibr" rid="ref6">6</xref>] play a central role in how sensitive medical data should be processed and protected. These regulatory requirements may pose substantial data protection and compliance challenges for the use of externally hosted LLMs with sensitive clinical data, shifting the focus for real-world clinical applications to locally operated LLMs that comply with data protection requirements by working behind the firewall of health care services. However, there has been little work to date focusing on these locally run LLMs. A study in head and neck cancer compared the performance of web-based and locally operated open-weight LLMs based on simulated cases [<xref ref-type="bibr" rid="ref7">7</xref>]. Auberville et al [<xref ref-type="bibr" rid="ref8">8</xref>] demonstrated that alignment methods, in-context learning, and parameter-efficient fine-tuning strongly increased the models&#x2019; overall performance on MDT recommendations, reaching congruence of up to 79% for the fine-tuned Mistral 7B.</p></sec><sec id="s1-5"><title>Objective of This Study</title><p>A review of the literature highlights the potential and limitations of applying LLMs in clinical practice [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref12">12</xref>]. However, promising technology can only make a difference if it is applied in clinical practice. Bridging the gap between simulation and practice, this feasibility study aims to provide the first prospective, real-world evaluation of guideline-instructed, nongrounded, locally operated LLMs as an augmentative tool for the regular MDT in head and neck cancer.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Ethics Approval</title><p>Ethical approval was obtained from the ethics committee of the state medical association (approval number 2024&#x2010;17946_2). The workflow of the study is illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Workflow of the study. LLM: large language models; MDT: multidisciplinary tumor boards.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95081_fig01.png"/></fig><p>Informed consent was obtained from patients with a malignant disease who were scheduled to be seen in the MDT for head and neck oncology at our clinic. Information from the regular MDT registration was presented to 2 different locally run LLMs using the prompt shown in <xref ref-type="fig" rid="figure2">Figure 2</xref> ahead of the MDT meeting. We evaluated 2 on-premises small open-weight models (gemma-3-12b and gpt-oss-20b) using a guideline-instructed prompt. The prompt requested guideline-based recommendations following a hierarchy (German S3 guideline as the primary reference, then the European Society for Medical Oncology, with the National Comprehensive Cancer Network only supplementary), but no document-level retrieval of guideline text was implemented; the models therefore relied on their internal knowledge when citing sources.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Prompt translated by DeepL (Cologne, Germany) [<xref ref-type="bibr" rid="ref13">13</xref>]. ENT: Ear Nose Throat; ESMO: European Society for Medical Oncology; NCCN: National Comprehensive Cancer Network.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95081_fig02.png"/></fig><p>The MDT meeting was conducted as usual, and MDT members, including specialists in otorhinolaryngology-head and neck surgery, oral and maxillofacial surgery, medical oncology, radiology, radiation oncology, and pathology, did not know in advance which cases were processed by an LLM. After the specific case was discussed within the usual workflow and a decision or recommendation was reached, the MDT was confronted with the recommendations of the 2 LLMs. The recommendations of the 2 LLMs were discussed, and, if applicable, the MDT recommendation was adjusted. This workflow was implemented due to ethical considerations, which were explicitly required by the ethics committee. The requirements stated that the MDT&#x2019;s decision should not be primarily influenced by the LLM, but merely optimized where possible. The ethics committee permitted corrections of the MDT recommendation only if the LLM raised a point that the MDT had, in fact, overlooked or misjudged. Furthermore, MDT members were asked independently to complete a short questionnaire. The questionnaire was anonymous, and responses could be submitted either digitally (LimeSurvey GmbH) or on paper. For each patient, a separate survey was conducted in which participants rated the recommendations of both LLMs separately. Participants could either complete the questionnaire straight away or after the MDT. The questions did not include any forced-choice options, so it was possible to submit partially completed questionnaires. Participants were asked to rate the medical adequacy of the 2 recommendations made by the respective LLMs on a 6-point Likert scale (1=&#x201C;very poor&#x201D; and 6=&#x201C;excellent&#x201D;). Moreover, MDT members were further asked whether the LLM&#x2019;s response could improve the board&#x2019;s decision (yes/no) and whether the respective LLM response was potentially hazardous for the patient (yes/no). &#x201C;Hazardous to patients&#x201D; was defined as the presence of information that could directly or indirectly cause harm to the patient if followed. This includes advice that contradicts established clinical guidelines, promotes unsafe practices, misrepresents risks or benefits, or could lead to delayed diagnosis, inappropriate treatment, or adverse outcomes. Additionally, MDT members were provided with the chance to give written feedback on the respective LLM recommendations. Beyond the rating of the MDT members, the therapy regimens of the MDT and the 2 LLMs were compared in a table for concordance.</p></sec><sec id="s2-2"><title>Outcomes</title><p>The primary outcome was the proportion of cases in which the MDT recommendation changed after review of the LLM outputs. Secondary outcomes were (1) concordance with MDT treatment intent (curative vs palliative), (2) overlap with the MDT first-line regimen, (3) MDT member ratings of medical adequacy, perceived potential benefit, and potential hazard, and (4) qualitative themes from free-text comments.</p></sec><sec id="s2-3"><title>LLM Execution</title><p>Information from the regular MDT registration, including diagnosis, tumor-node-metastasis stage, date of initial diagnosis, imaging reports, pathology reports, previous treatment, secondary diagnoses, comments, and the query to the MDT, was transferred to a Word document (Microsoft Word) and presented to the nongrounded, locally run LLMs using the prompt shown in <xref ref-type="fig" rid="figure2">Figure 2</xref> (for the full prompt, see <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). LLM results were retrieved based on a single-shot strategy. The LLMs were run locally on a standard Hewlett Packard (Palo Alto, CA, USA) notebook (Intel Core i7-1255U, 4.7 GHz; DDR4, 16 GB [2&#x00D7;8 GB], Windows 10 Pro; LM Studio version 0.3.15/23). The selection of LLMs was guided by their open-weight availability and model size. Consequently, only models with below 13 GB were included. The default settings of LM Studio were not modified. All model configurations&#x2014;including parameter sizes or variants, quantization settings, prompt templates and system prompts, as well as temperature, top-p, maximum token limits, and random seeds&#x2014;were kept unchanged. The specific models evaluated in this study are as follows: lmstudio-community/gemma-3-12b-it-gguf/gemma-3-12b-it-q3_k_l.gguf (GPU offload 33/42; CPU thread pool size 4; evaluation batch size 512; temperature 0.1; top-<italic>p</italic> sampling 0.95; context length 4096; random seed) and lmstudio-community/gpt-oss-20b-gguf/gpt-oss-20b-mxfp4.gguf (GPU offload 11/40; CPU thread pool size 4; evaluation batch size 512; temperature 0.8; top-<italic>p</italic> sampling 0.95; context length 4096; random seed; all published by lmstudio-community).</p></sec><sec id="s2-4"><title>Statistical Analysis</title><p>All responses from the raters were transferred to an Excel spreadsheet (Microsoft Excel) and sorted according to the entity being evaluated. The statistical analysis was performed using GraphPad Prism software (version 10.6.1 for macOS; GraphPad Software). The data did not meet normality assumptions, as confirmed by the D&#x2019;Agostino and Pearson test (ns: <italic>P</italic>&#x003E;.05, *<italic>P</italic>&#x003C;.05, **<italic>P</italic>&#x003C;.005, ***<italic>P</italic>&#x003C;.0005). Differences regarding the ratings for medical adequacy (6-point Likert scale) between LLMs were analyzed by the Wilcoxon matched-pairs signed-rank test. The word count among different LLM answers was analyzed by the Friedman test and Dunn multiple comparison post hoc test. Binary-rated categories were analyzed for significant differences using exact McNemar test in SAS software (version 9.4 for Windows; SAS Institute Inc). The 95% CIs of proportions were calculated by the Clopper-Pearson method using the Ausvet 2026 Epitools CI Calculator [<xref ref-type="bibr" rid="ref14">14</xref>].</p><p>MDT members completed anonymous questionnaires independently, either digitally or on paper. As participation varied between cases and questionnaires did not contain mandatory fields, the number of ratings differed across patients and models, resulting in an incomplete and unbalanced dataset without persistent rater identifiers. Therefore, formal interrater reliability statistics requiring fixed rater assignments across items (eg, Fleiss &#x03BA; or the intraclass correlation coefficient) were considered methodologically inappropriate. Instead, rating consistency was assessed descriptively for each patient case and LLM separately using the number of ratings, median, IQR, range, and the proportion of the most frequent rating. Analysis and data processing for this part of the analysis were performed using Python in Google Colab to ease the implementation.</p></sec><sec id="s2-5"><title>Qualitative Rating by MDT Members and Post Hoc Analysis of LLMs Recommendations</title><p>The comments of the MDT members were collected in a Word document and summarized. Additionally, the authors&#x2019; observations during the tabular comparison of concordance in therapy regimens were also summarized and kept on record.</p></sec><sec id="s2-6"><title>Data Privacy</title><p>All data were processed on locally run entities, preserving the data privacy of patients. The data underlying this paper cannot be shared publicly to protect the privacy of the individuals who participated in the study.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Relevant Amendment of MDT Recommendation</title><p>In 1 case, the LLM (gemma-3-12b) pointed out a secondary finding that had been noticed during staging (computed tomography scan) but had not yet been checked. In this case, the MDT&#x2019;s recommendation was substantially amended with a corresponding note regarding a follow-up examination.</p></sec><sec id="s3-2"><title>Concordance in Therapy Regimen</title><p>Recommended therapy regimens generated by the MDT and the different LLMs are illustrated in <xref ref-type="table" rid="table1">Tables 1</xref> and <xref ref-type="table" rid="table2">2</xref>. While the MDT recommended a curative treatment regimen for 88% (15/17, 95% CI 64%&#x2010;99%) of patients, gemma-3-12b recommended this for 82% (14/17, 95% CI 57%&#x2010;96%), and gpt-oss-20b for 59% (10/17, 95% CI 33%&#x2010;82%). This corresponds to 94% (16/17, 95% CI 71%&#x2010;100%) agreement between gemma-3-12b and the MDT and a 59% (10/17, 95% CI 33%&#x2010;82%) agreement between gpt-oss-20b and the MDT. For 35% (6/17, 95% CI 14%&#x2010;62%) of patients, gemma-3-12b stated the same first-line therapy regimen as the MDT as first-line, with gpt-oss-20b achieving the same for 41% (7/17, 95% CI 18%&#x2010;67%) of patients. In 59% (10/17, 95% CI 33%&#x2010;82%) of patients, gemma-3-12b stated the MDT&#x2019;s first-line therapy regimen, albeit with a different priority, while in gpt-oss-20b, it was 41% (7/17, 95% CI 18%&#x2010;67%) of patients. For a further 24% (4/17, 95% CI 7%&#x2010;50%) of patients, gemma-3-12b and for 47% (8/17, 95% CI 23%&#x2010;72%) of patients, gpt-oss-20b stated parts of the first-line therapy regimen recommended by the MDT.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Comparison of curative and palliative therapy regimens (N=17).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top"/><td align="left" valign="top">MDT<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>, n (%; 95% CI)</td><td align="left" valign="top" colspan="2">gemma-3-12b, n (%; 95% CI)</td><td align="left" valign="top" colspan="2">gpt-oss-20b, n (%; 95% CI)</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Recommendation</td><td align="left" valign="top">Concordance with MDT</td><td align="left" valign="top">Recommendation</td><td align="left" valign="top">Concordance with MDT</td></tr></thead><tbody><tr><td align="left" valign="top">Therapy regimen</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">16 (94; 71&#x2010;100)</td><td align="left" valign="top"/><td align="left" valign="top">10 (59; 33&#x2010;82)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Curative therapy regimen</td><td align="left" valign="top">15 (88; 64&#x2010;99)</td><td align="left" valign="top">14 (82; 57&#x2010;96)</td><td align="left" valign="top"/><td align="left" valign="top">10 (59; 33&#x2010;82)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Palliative therapy regimen</td><td align="left" valign="top">2 (12; 1&#x2010;36)</td><td align="left" valign="top">3 (18; 4&#x2010;43)</td><td align="left" valign="top"/><td align="left" valign="top">0 (0; 0&#x2010;20)</td><td align="left" valign="top"/></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>MDT: multidisciplinary tumor board. *For 7 cases, gpt-oss-20b provided no clear treatment recommendation and instead suggested additional examinations.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>First-line therapy regimens recommended by the large language models compared with the multidisciplinary tumor board (N=17).</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Recommendation</td><td align="left" valign="bottom">gemma-3-12b, n (%; 95% CI)</td><td align="left" valign="bottom">gpt-oss-20b, n (%; 95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">Stated all first-line therapy regimens of the MDT<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="char" char="." valign="top">10 (59; 33&#x2010;82)</td><td align="char" char="." valign="top">7 (41; 18&#x2010;67)</td></tr><tr><td align="left" valign="top">Stated some first-line therapy regimens of the MDT as first-line</td><td align="char" char="." valign="top">4 (24; 7&#x2010;50)</td><td align="char" char="." valign="top">8 (47; 23&#x2010;72)</td></tr><tr><td align="left" valign="top">Stated all first-line therapy regimens as first-line, consistent with the MDT</td><td align="char" char="." valign="top">6 (35; 14&#x2010;62)</td><td align="char" char="." valign="top">7 (41; 18&#x2010;67)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>MDT: multidisciplinary tumor board.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Quantitative Rating by MDT Members</title><p>Medical adequacy, as rated by the MDT members, revealed a median rating of 5 (IQR 2&#x2010;5) for gemma-3-12b and a median of 4 (IQR 3&#x2010;5) for gpt-oss-20b (<xref ref-type="fig" rid="figure3">Figure 3A</xref>). The MDT members stated that the information provided by gemma-3-12b had the potential to improve the MDT decision in 30% (28/93, 95% CI 21%-40%) of ratings. For gpt-oss-20b, this was stated in 23% (19/83, 95% CI 14%&#x2010;33%) of ratings. Regarding potential hazards for patients, MDT members stated potentially hazardous information in 27% (25/93, 95% CI 18%-37%) of ratings for gemma-3-12b and 17% (14/83, 95% CI 9%-26%) of ratings for gpt-oss-20b. No significant difference (<italic>P=0.8094</italic>) between the LLMs was found for medical adequacy according to the Wilcoxon test (<xref ref-type="fig" rid="figure3">Figure 3A</xref>). Regarding improvement of the board recommendation (<xref ref-type="fig" rid="figure3">Figure 3B</xref>), no significant difference was observed between the LLMs (exact McNemar test, P=0.0707). In contrast, a statistically significant difference was observed for potentially hazardous information (<xref ref-type="fig" rid="figure3">Figure 3C</xref>), with hazardous information reported in 27% (25/93) of ratings for gemma-3-12b compared with 17% (14/83) for gpt-oss-20b (exact McNemar test, P=0.0330)t.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Quantitative rating by multidisciplinary tumor board members. Medical adequacy rated on a 6-point Likert scale (A), shown as a box plot. Improvement of the board recommendation (B) and presence of information potentially hazardous for patients (C), shown as binary outcomes in absolute numbers (n). Differences between the LLMs were tested using the Wilcoxon test for medical adequacy (ns; <italic>P=0.8094</italic>) and the exact McNemar test for &#x2018;Improvement in the quality of the board recommendation&#x2019; (B) and &#x2018;Potentially hazardous information&#x2019; (C). The difference was not statistically significant for improvement of the board recommendation (P=0.0707) but was statistically significant for potentially hazardous information (P=0.0330).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95081_fig03.png"/></fig></sec><sec id="s3-4"><title>Qualitative Rating by MDT Members and Post Hoc Analysis of LLMs Recommendations</title><p>The LLM integration was beneficial in verifying the completeness of MDT submissions. However, some recommendations of the LLMs also indicated potential hazards for patients. Therefore, the free-text comments of MDT members indicating potential hazards were systematically categorized into factual errors, misinterpretation of the case, reasoning errors, and semantic errors (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). For both LLMs, the MDT members stated factual errors for 4 patients each. Misinterpretation of the case was found in 3 cases for gemma-3-12b and 1 case for gpt-oss-20b. Reasoning errors were highlighted in 2 cases for gemma-3-12b and 1 case for gpt-oss-20b. Moreover, the MDT emphasized semantic errors in 1 case for gpt-oss-20b and none for gemma-3-12b.</p><p>Factual errors often concerned decisions regarding adjuvant treatment. In cases where the LLMs&#x2019; recommendations differed from the MDT&#x2019;s decisions, both directions were present: either no recommendation for adjuvant therapy despite the presence of risk factors (undertreatment) or a recommendation for adjuvant therapy in the absence of risk factors (overtreatment).</p><p>Although general information, such as the reference to resection margins, was presented to the LLMs, in some cases the LLMs lost track of the full clinical picture. For example, sample collection by panendoscopy was classified as an R1 resection, or a tonsillectomy performed as part of a diagnostic panendoscopy was classified as a curatively intended resection of the tumor.</p><p>However, MDT members frequently felt that the reasoning behind the LLMs&#x2019; treatment recommendations was incorrect. In one example, the indication for radiotherapy was based solely on the resection status, even though the indication was already present due to the tumor (T) stage. Furthermore, the LLMs sometimes did not include important information, such as prior exposure to radiotherapy, in the decision-making process, even though this information was evident from the documents provided.</p><p>In some cases, new words were created by the LLMs, such as the description of a &#x201C;pulmocarcinoma.&#x201D; There was also confusion regarding the distinction between neoadjuvant, primary, and adjuvant radio(chemo)therapy. In some cases, the therapy recommended by the LLM could be deduced from the context, but in others it remained unclear what the LLM was exactly referring to. The LLMs seemed to lose track, particularly in complex cases where additional tumors were present, such as a parallel secondary carcinoma of the lung.</p><p>It was also noticeable that gpt-oss-20b showed a marked preference for PET-CTs (positron-emission-tomography-computed tomography). While a PET-CT was performed before the MDT in only 1 case, and a PET-CT was recommended by the MDT as the next diagnostic step for another patient, gpt-oss-20b mentioned PET-CT for 8 patients in its recommendation. In a further 5 patients, gpt-oss-20b insisted on performing a PET-CT scan before providing further recommendations. In only 4 patients did gpt-oss-20b not mention a PET-CT scan at all. In contrast, gemma-3-12b only considered a PET-CT scan in 3 patients. In one of these patients, the MDT recommended the same, and another patient received a PET-CT ahead of the MDT. Some MDT members criticized the fact that the LLM (gpt-oss-20b) frequently required a PET-CT scan and did not discuss any further treatment steps. It was pointed out that the LLMs should provide treatment plans for different investigation outcome scenarios in order to reduce further delay in treatment.</p></sec><sec id="s3-5"><title>Number of Words</title><p>The word count of the recommendations provided by the MDT and the LLMs is visualized in <xref ref-type="fig" rid="figure4">Figure 4</xref>. While gemma-3-12b showed the highest word count, with a median of 256 (IQR 233.5&#x2010;278) words, gpt-oss-20b used a median of 105 (IQR 69&#x2010;122) words, and the MDT a median of 20 (IQR 14.5&#x2010;30) words. Intergroup comparison in the Friedman test showed significant differences (<italic>P</italic>&#x003C;.001) among groups. The post hoc analysis using Dunn multiple comparison test revealed significant differences between the MDT and the gemma-3-12b (<italic>P</italic>&#x003C;.001), as well as between gemma-3-12b and gpt-oss-20b (<italic>P</italic>&#x003C;.005). No significant differences were found between the MDT and gpt-oss-20b (<italic>P</italic>&#x003E;.05).</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Word count of the recommendations stated by the MDT, gemma-3-12b, and gpt-oss-20b. A Friedman test showed significant differences in word count between the entities (<italic>P</italic>&#x003C;.001). Post hoc analysis was performed using the Dunn multiple comparison test (ns: <italic>P</italic>&#x003E;.05; **<italic>P</italic>&#x003C;.005; ****<italic>P</italic>&#x003C;.001). MDT: multidisciplinary tumor board.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95081_fig04.png"/></fig></sec><sec id="s3-6"><title>IDescriptive Analysis of Interrater Agreement</title><p>The number of MDT members ranged from 8 to 12 (median 11, IQR 11&#x2010;12), and the LLMs received a median number of 6 (IQR 4&#x2010;6) ratings per case. The within-case descriptive agreement analysis demonstrated heterogeneous rating dispersion across cases and models. Several cases showed narrow IQRs and high proportions of identical ratings, indicating substantial agreement among MDT participants. However, other cases demonstrated broader rating distributions and wider ranges, reflecting greater variability in expert assessment. Overall, the analysis suggested that agreement was case-dependent rather than uniformly high or low across all evaluations. For detailed information, see Table S2 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> (within-case variability of expert Likert ratings).</p></sec><sec id="s3-7"><title>Implications for Further Studies</title><p>A power analysis was carried out using the exact McNemar test to determine the sample sizes required for future studies. Based on 80% power, a total of 240 patients (5% vs 13% discordant pairs) would be required to detect an improvement in the board recommendation outcome. Regarding the rating of potential hazard to patients, a total of 150 patients (7% vs 19% discordant pairs) would be required to achieve 80% power at a 5% significance level.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Contextualization</title><p>Although the application of LLMs for MDT augmentation has been extensively evaluated, previous monocenter simulation (in silico) studies in head and neck cancer focused primarily on web-based LLMs [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Only 2 studies tested the performance of locally run LLMs in head and neck cancer MDTs. However, these studies used constructed cases rather than real-world data [<xref ref-type="bibr" rid="ref7">7</xref>] or had a retrospective design [<xref ref-type="bibr" rid="ref8">8</xref>]. Accordingly, this is the first prospective study of its kind to use real-world MDT augmentation by guideline-instructed, nongrounded, locally operated LLMs in the field of head and neck oncology. Much of the current literature remains in silico. LLMs are evaluated on simulated vignettes or retrospectively curated cases, often with more complete information than is available at the time of a live MDT. By contrast, real-world prospective testing captures practical constraints such as incomplete referrals, time pressure, and the need for short, actionable outputs, and therefore provides a more stringent assessment of clinical utility and risk.</p></sec><sec id="s4-2"><title>Performance of Locally Operated LLMs</title><p>Within this study, gemma-3-12b showed 94% (16/17, 95% CI 71%&#x2010;100%) and gpt-oss-20b showed 59% (10/17, 95% CI 33%&#x2010;82%) concordance with the MDT regarding the curative or palliative therapy regimen (<xref ref-type="table" rid="table1">Table 1</xref>). For 7 cases of gpt-oss-20b no clear recommendation was provided and additional examination was suggested. In similar preliminary work with simulated cases in the head and neck region, 92% concordance was reported for Llama 3 and 84% concordance for ChatGPT-4o [<xref ref-type="bibr" rid="ref7">7</xref>]. Accordingly, Gemma-3-12b achieves slightly higher concordance values than those reported in the literature, while gpt-oss-20b achieves lower values. Albeit with a different priority, gemma-3-12b stated the same first-line therapy regimen as the MDT in 59% (10/17, 95% CI 33%&#x2010;82%) of cases, while gpt-oss-20b reached 41% (7/17, 95% CI 18%&#x2010;67%) concordance with the MDT. Our previous study evaluating constructed cases showed 64% concordance for ChatGPT-4o and 60% for Llama 3 [<xref ref-type="bibr" rid="ref7">7</xref>]. In 35% (6/17, 95% CI 14%&#x2010;62%) of cases, gemma-3-12b stated the same first-line therapy regimen as the MDT as first-line therapy, whereas gpt-oss-20b achieved 41% (7/17, 95% CI 18%&#x2010;67%) concordance (<xref ref-type="table" rid="table1">Table 1</xref>). Here, both LLMs tested in this study performed worse than the published performance of ChatGPT-4o (52%) and Llama 3 (48%) [<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>Regarding medical adequacy, gemma-3-12b reached a median rating of 5 (IQR 2&#x2010;5) and gpt-oss-20b a median of 4 (IQR 3&#x2010;5) in this study. Here, no significant difference (<italic>P</italic>&#x003E;.05) between the LLMs was found (<xref ref-type="fig" rid="figure3">Figure 3A</xref>). While gemma-3-12b outperformed published values of ChatGPT-4o (median 4.7, IQR 4&#x2010;6) and Llama 3 (median 4.3, IQR 3&#x2010;5), gpt-oss-20b underperformed compared with the published benchmark [<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>MDT members found the LLMs to have the potential to improve MDT decisions in 30% (28/93, 95% CI 21%&#x2010;40%) for gemma-3-12b and 23% (19/83, 95% CI 14%&#x2010;33%) for gpt-oss-20b (<xref ref-type="fig" rid="figure3">Figure 3B</xref>). Here, in the previous study, the MDT members stated the same in 17% of ratings [<xref ref-type="bibr" rid="ref7">7</xref>]. With regard to potential hazards for patients, MDT members reported potentially hazardous information in 27% (25/93, 95% CI 18%&#x2010;37%) of ratings for gemma-3-12b and 17% (14/83, 95% CI 9%&#x2010;26%) of ratings for gpt-oss-20b. This difference was statistically significant (exact McNemar test, P=0.0330) highlighting that differences between locally run LLMs may not only concern the quality or completeness of their recommendations but also their safety profile.</p><p>Within the prompt, both LLMs were instructed to limit their response to 100 words (<xref ref-type="fig" rid="figure2">Figure 2</xref>). While gpt-oss-20b adhered to the word limit fairly well (median 105, IQR 69&#x2010;122), gemma-3-12b deviated from it (median 256, IQR 233.5&#x2010;278) and showed a significant deviation (<italic>P</italic>&#x003C;.001) from the MDT&#x2019;s response length (median 20, IQR 14.5&#x2010;30; <xref ref-type="fig" rid="figure4">Figure 4</xref>).</p></sec><sec id="s4-3"><title>Qualitative Performance of the LLMs</title><p>The qualitative analysis of the LLM recommendations highlights the current inability of the tested LLMs to generate comprehensive therapy recommendations similar to those of an MDT. The MDT members felt the LLMs offered incorrect justifications for therapy recommendations, confusion regarding nomenclature (eg, in distinguishing between neoadjuvant, primary, and adjuvant radio(chemo)therapy), and the misunderstanding that a panendoscopy with sample collection does not yet include surgical resection. Furthermore, the LLMs created surprising neologisms such as &#x201C;pulmocarcinoma.&#x201D; These errors can be considered as a form of LLM &#x201C;hallucination&#x201D; [<xref ref-type="bibr" rid="ref15">15</xref>]. Although gpt-oss-20b has a reasoning mechanism, these errors occurred in both LLMs. Such neologisms arise because mid-sized local LLMs recombine meaningful medical morphemes when they lack strong domain-specific lexical constraints. This reflects limited specialization and alignment rather than random error.</p><p>Another phenomenon was the tendency of the LLMs to suggest further examinations. For instance, gpt-oss-20b recommended PET-CT for multiple patients. This phenomenon was further underscored by the fact that gpt-oss-20b never recommended palliative treatment and instead suggested to broaden the diagnostic scope. The tendency of LLMs to demand more diagnostics than doctors has been described in previous work [<xref ref-type="bibr" rid="ref1">1</xref>] and might be justified by their focus on safety. Nonetheless, additional examinations are accompanied by risks such as radiation exposure or delayed time to treatment and incur relevant resources. This behavior stems from LLMs&#x2019; being trained to maximize uncertainty reduction and patient safety, which biases them toward recommending additional diagnostic tests. Unlike clinicians, the models do not directly incorporate real-world constraints such as workflow pressure, resource availability, and patient-specific feasibility unless these are explicitly represented in the input.</p></sec><sec id="s4-4"><title>Analysis of Interrater Agreement</title><p>The observed variability in expert ratings likely reflects the inherent complexity and interpretative nature of multidisciplinary oncological decision-making. While some LLM-generated recommendations were evaluated consistently by MDT participants, other cases elicited broader disagreement, potentially reflecting differing clinical perspectives among specialties. Due to the anonymous and incomplete survey structure, formal inter-rater reliability statistics could not be robustly applied. Nevertheless, the case-based descriptive agreement analysis provides transparent insight into the consistency and dispersion of expert evaluations within the study cohort.</p></sec><sec id="s4-5"><title>Implications of the Study</title><p>To our knowledge, there are no studies investigating the prospective use of locally operated LLMs in head and neck cancer. The present study evaluates the feasibility of integrating locally operated LLMs in today&#x2019;s MDT process. In clinical practice, MDT submissions are checked for completeness and accuracy by a doctor before every MDT. Within our study, this preprocessing took around 15 minutes per case, including processing by both tested LLMs. Accordingly, the extra time required is only a few minutes per case. This also highlights a positive aspect of LLM integration: LLMs can assist during this process, for example, by helping to identify inconsistencies in the submissions or pointing out missing data or diagnostic information. The presentation and discussion of the LLMs&#x2019; recommendations within the MDT took only a few minutes per case. This workflow was appreciated by the participants and received with interest. However, in this study, we have limited the discussion to the outputs of two LLM cases in order to avoid an overload of the MDT meeting.</p><p>Presenting the LLM outputs after the MDT decision may introduce some degree of bias. In this study, this specific workflow was required due to ethical considerations, as the MDT&#x2019;s decision should not be primarily influenced by the LLMs. One could argue that, given the sequence of events, doctors might find it more difficult to deviate from the decision that has been made. However, within our study, the recommendation was adjusted by the LLM in 1 case, as mentioned above. On the contrary, the applied order reduces the risk of automation bias because the MDT finds consensus before knowing the LLMs&#x2019; recommendation. Automation bias occurs when doctors rely too heavily on AI decisions [<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>Evaluating the models using a single-shot strategy, their respective default LM Studio inference configurations and quantization schemes, and default settings is consistent with a real-world setup. However, differences in temperature and quantization may have influenced output determinism and reasoning performance. Therefore, results should be interpreted within the context of practical deployment settings rather than fully standardized inference conditions.</p><p>The statistical analyses did not explicitly account for clustering, as multiple ratings were obtained per patient case and LLM response. Because ratings were collected anonymously without persistent reviewer identifiers, retrospective modeling of rater-level clustering using mixed-effects models or generalized estimating equations was not feasible. This may have affected the precision of <italic>P</italic> values and confidence intervals. Therefore, the reported inferential statistics should be interpreted with appropriate caution.</p><p>We deliberately evaluated 2 small open-weight models (12B and 20B parameters) that can run on standard local hardware to maximize feasibility, decentralization, and data protection by keeping all processing behind the institutional firewall. This design contrasts with cloud-based, full-size proprietary LLMs (&#x201C;cloud full LLMs&#x201D;), which may offer higher performance but typically require the transfer of patient data to an external provider.</p><p>An intermediate deployment option is large open-weight models self-hosted on institutional GPU servers. Such models may narrow the performance gap with cloud-based systems while maintaining on-premises data control, but they increase infrastructure requirements and may be less accessible for smaller centers. Checking for completeness also proved to be a major advantage of the LLMs, showcasing a promising use case for LLM implementation. Here, LLMs supporting the MDT registration process can identify missing examinations prior to the MDT, avoiding unnecessary delays in therapy [<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>Although the treatment regimens between gemma-3-12b and the MDT matched in 35% (6/17, 95% CI 14%&#x2010;62%) of patients and between gpt-oss-20b and MDT in 41% (7/17, 95% CI 18%&#x2010;67%) of patients, the recommendations of the LLMs were not able to replace the MDT recommendation in these patients. The reasons for this are various minor errors in the justification for the therapy or an incorrect summary of the case (panendoscopy with sample collection being considered a surgical resection).</p><p>Nevertheless, MDT members found the recommendations of the LLMs helpful in 30% (28/93, 95% CI 21%&#x2010;40%) for gemma-3-12b and 23% (19/83, 95% CI 14%&#x2010;33%) for gpt-oss-20b. This suggests that carefully supervised real-world implementation of LLMs may provide useful adjunctive support in MDT workflows, rather than augmenting or replacing the MDT&#x2019;s opinion.</p><p>Larger LLMs running on large servers are likely to provide improved results, and further development and improvement of the models may additionally lead to more suitable recommendations. Other enhancement opportunities are fine-tuning [<xref ref-type="bibr" rid="ref8">8</xref>] and grounding the LLMs based on relevant guidelines and current studies [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. Importantly, our approach was guideline-instructed rather than guideline-grounded. The LLMs were prompted to follow a guideline hierarchy and to cite sources, but were not provided with the guideline documents themselves. Guideline-grounded systems that retrieve and supply the relevant guideline passages at inference time could improve factuality, enable auditable citations, and reduce hallucinated or outdated recommendations. The frequency of potentially hazardous content observed in this study supports prioritizing such grounding in future implementations. Both approaches should be further evaluated in future studies in head and neck. Ideally, an &#x201C;in the loop&#x201D; process is being established in which LLMs are applied, supervised by doctors, and continuously optimized in appropriate centers.</p><p>Further improvements may be achieved through domain-specific vocabulary constraints, retrieval-augmented generation to integrate up-to-date guidelines and clinical evidence at inference time [<xref ref-type="bibr" rid="ref19">19</xref>], and recursive language models enabling iterative refinement of clinical reasoning [<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>Rating medical adequacy, potential improvement, and hazard for patients can highly be affected inter alia by the personal views, experience, and last but not least, the subspecialty of the raters (MDT members). However, this is a general problem of evaluating LLMs. In order to address this issue, we analyzed the LLMs&#x2019; recommendations for concordance with the MDT recommendation. The MDT found consensus before knowing the LLMs&#x2019; recommendation to avoid influence. The MDT itself is an organ of internal validation as interdisciplinary perspectives on treatment are bundled in a meeting to combine the expertise and validate the treatment approach. Furthermore, our university hospital is a German Cancer Society&#x2013;validated tumor center and thus subject to external validation on a regular basis. The comparison for concordance with the MDT was implemented in order to provide a more objective assessment. However, even the definition of concordance is prone to bias of subjectivity. Our study was specifically designed to test a use case that could be implemented straightforwardly in a clinical setting. The more complex the issue and the case, the more complex the assessment. When it comes to cancer patients&#x2019; treatment plans, there might be more than a single correct recommendation, despite the existence of oncological guidelines. Treatment strategies depend on a great variety of different nuances, due to risk factors, comorbidities, and the patient&#x2019;s preferences, as well as the center&#x2019;s expertise or regional specifications. Other use cases for LLMs, such as determining clearly defined tumor stages based on imaging findings, are significantly easier to evaluate [<xref ref-type="bibr" rid="ref21">21</xref>]. As the perception of LLMs is also highly relevant, we have deliberately opted for a subjective criterion assessed by the MDT participants (experts) and a comparison with a more objective criterion: the MDT&#x2019;s recommendation (the current gold standard). At present, we consider this approach to be sensible also for other specialist fields beyond otorhinolaryngology-head and neck surgery. This may change in the future. The evaluation of LLMs themselves also offers plenty of scope for future research.</p></sec><sec id="s4-6"><title>Ethical Considerations</title><p>An application of AI in clinical practice is subject to ethical discussions. Despite the obvious shortcomings of the LLMs in this study, in one specific case, the LLM made a relevant difference by pointing out that a further examination was pending. This examination might have been overlooked without the LLM&#x2019;s recommendation. This instance raises a central question: Is it ethically justifiable not to use AI in the clinic, even if&#x2014;as in this study&#x2014;it only makes a relevant difference for one patient? Of course, it is not a question of replacing the doctor; the local LLMs in this study have proven that they are not capable of replacing human medical recommendations. Beyond measurable criteria for decision-making, being a doctor includes interpersonal relationships and the &#x201C;soft parameters,&#x201D; which cannot be replaced by AI. However, perhaps AI should be implemented as an additional tool in general, even if, at the current stage, it is only to check for completeness, which may not always be checked with absolute accuracy in the hurry of everyday clinical practice.</p></sec><sec id="s4-7"><title>Conclusion</title><p>This first prospective study demonstrates the feasibility of an auxiliary implementation of LLMs in today&#x2019;s MDT&#x2019;s workflow. The low-barrier setup using small, locally run, open-weight LLMs suitable for standard computers can have a positive influence on MDT decisions in individual cases. However, rather than substantial improvement of MDT decisions, the benefits of the LLMs&#x00B4; recommendations were mainly limited to the indication of follow-up examinations that are at risk of being lost within the flood of information. Nonetheless, implementation of LLMs as an augmentation of MDT decision-making is still promising. First, even occasional, and seemingly modest improvements of MDT recommendations, like the reminder regarding a pending follow-up examination, could significantly influence the fate of a specific patient. Second, the performance of (locally run) LLMs will further improve in the foreseeable future. More advanced models will very likely offer further improvements in more aspects of recommendation. Beyond checking the completeness of the provided data and planned examinations, LLMs may support the development of individualized treatment concepts based on recent studies that have not yet been incorporated into clinical guidelines. Furthermore, suitable LLMs may streamline patients&#x2019; inclusion in appropriate prospective experimental treatment studies [<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref24">24</xref>]. Future studies should focus on the use of locally operated LLMs with the potential data protection advantages of local deployment and greater likelihood of being implemented within existing clinical IT infrastructure. Although multiple retrospective studies evaluating the use of LLMs in MDTs have been published, prospective studies, particularly those evaluating locally run LLMs, remain limited. The present study therefore contributes to the emerging evidence on prospective implementation of locally run LLMs and aims to pave the way for pragmatic implementation in real clinical practice.</p></sec></sec></body><back><ack><p><xref ref-type="fig" rid="figure1">Figures 1</xref> and <xref ref-type="fig" rid="figure2">2</xref> were drawn by JE and CRB using Microsoft PowerPoint. <xref ref-type="fig" rid="figure3">Figures 3</xref> and <xref ref-type="fig" rid="figure4">4</xref> were assembled by JE and CRB using Prism for Windows (version 9.5.1; GraphPad Software).</p><p>The authors declare the use of generative AI during the publication and revision process, including journal selection, research support, manuscript editing, and statistical code development. According to the GAIDeT taxonomy (2025), the following tasks were delegated to generative AI tools under full human supervision:</p><p>Evaluation of the novelty of the research and identification of gaps</p><p>Code generation</p><p>Code optimization</p><p>Creation of algorithms for data analysis</p><p>Proofreading and editing</p><p>Translation</p><p>Quality assessment</p><p>Publication support</p><p>The AI tools used were DeepL (DeepL SE) for translation from German to English (because the authors are nonnative English speakers) and ChatGPT (OpenAI), using the default model available to free users at the time of use (May 2025). All AI-generated outputs were critically reviewed, verified, and, where necessary, revised by the authors. Responsibility for the final manuscript lies entirely with the authors. Generative AI tools are not listed as authors and do not bear responsibility for the final manuscript.</p><p>Declaration submitted by: CRB.</p></ack><notes><sec><title>Funding</title><p>This study was funded by internal resources. During the study, the first author, CRB, was a TransMed Fellow, an internal funding program for clinician scientists at Mainz University Medical Center.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: CRB (lead), LM, DPS, DT, JJP, AB, CR, CM, SK, JE (co-lead)</p><p>Data curation: CRB (lead), JJP, CR, JE (co-lead)</p><p>Formal analysis: CRB (lead), LM, DPS, DT, JJP, MK, KO, JK, HN, MG, SZ, SA, AB, CR, CM, SK, JE (co-lead)</p><p>Investigation: CRB (lead), LM, DPS, DT, JJP, MK, KO, JK, HN, MG, SZ, SA, CM</p><p>Methodology: CRB, LM, DPS, DT, JJP, AB, CR, CM, SK, JE (co-lead)</p><p>Project administration: CRB (lead), JE (co-lead)</p><p>Resources: CRB, LM, DPS, DT, JJP, MK, KO, JK, HN, MG, SZ, SA, AB, CR, CM, SK, JE</p><p>Supervision: CRB (lead), SK, JE (co-lead)</p><p>Validation: CRB, LM, DPS, DT, JJP, MK, KO, JK, HN, MG, SZ, SA, CR, CM, JE</p><p>Visualization: CRB, LM, DPS, DT, JJP, MK, KO, JK, HN, MG, SZ, SA, AB, CR, CM, SK, JE</p><p>Writing &#x2013; original draft: CRB (lead), LM, DPS, DT, JJP, MK, KO, JK, HN, MG, SZ, SA, AB, CR, CM, SK, JE (co-lead)</p><p>Writing &#x2013; review &#x0026; editing: CRB (lead), LM, DPS, DT, JJP, MK, KO, JK, HN, MG, SZ, SA, AB, CR, CM, SK, JE (co-lead)</p></fn><fn fn-type="conflict"><p>SK is a founder &#x0026; shareholder of MED.digital. All other authors declare no conflicts of interest.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">GPU</term><def><p>graphics processing unit</p></def></def-item><def-item><term id="abb2">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb3">MDT</term><def><p>multidisciplinary tumor board</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lechien</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Chiesa-Estomba</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Baudouin</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hans</surname><given-names>S</given-names> </name></person-group><article-title>Accuracy of ChatGPT in head and neck oncological board decisions: preliminary findings</article-title><source>Eur Arch Otorhinolaryngol</source><year>2024</year><month>04</month><volume>281</volume><issue>4</issue><fpage>2105</fpage><lpage>2114</lpage><pub-id pub-id-type="doi">10.1007/s00405-023-08326-w</pub-id><pub-id pub-id-type="medline">37991498</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmidl</surname><given-names>B</given-names> </name><name name-style="western"><surname>H&#x00FC;tten</surname><given-names>T</given-names> </name><name name-style="western"><surname>Pigorsch</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Assessing the role of advanced artificial intelligence as a tool in multidisciplinary tumor board decision-making for primary head and neck cancer cases</article-title><source>Front Oncol</source><year>2024</year><volume>14</volume><fpage>1353031</fpage><pub-id pub-id-type="doi">10.3389/fonc.2024.1353031</pub-id><pub-id pub-id-type="medline">38854718</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmidl</surname><given-names>B</given-names> </name><name name-style="western"><surname>H&#x00FC;tten</surname><given-names>T</given-names> </name><name name-style="western"><surname>Pigorsch</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Assessing the use of the novel tool Claude 3 in comparison to ChatGPT 4.0 as an artificial intelligence tool in the diagnosis and therapy of primary head and neck cancer cases</article-title><source>Eur Arch Otorhinolaryngol</source><year>2024</year><month>11</month><volume>281</volume><issue>11</issue><fpage>6099</fpage><lpage>6109</lpage><pub-id pub-id-type="doi">10.1007/s00405-024-08828-1</pub-id><pub-id pub-id-type="medline">39112556</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vural Camalan</surname><given-names>B</given-names> </name><name name-style="western"><surname>Doluoglu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Taraf</surname><given-names>NH</given-names> </name><name name-style="western"><surname>Gunay</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Ozlugedik</surname><given-names>S</given-names> </name></person-group><article-title>ChatGPT versus DeepSeek in head and neck cancer staging and treatment planning: guideline-based study</article-title><source>Eur Arch Otorhinolaryngol</source><year>2025</year><month>09</month><volume>282</volume><issue>9</issue><fpage>4815</fpage><lpage>4824</lpage><pub-id pub-id-type="doi">10.1007/s00405-025-09524-4</pub-id><pub-id pub-id-type="medline">40523995</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="report"><article-title>Health Insurance Portability and Accountability Act of 1996</article-title><year>1996</year><access-date>2026-07-28</access-date><publisher-name>U.S. Government Publishing Office (GPO) / U.S. Congress</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.congress.gov/104/plaws/publ191/PLAW-104publ191.pdf">https://www.congress.gov/104/plaws/publ191/PLAW-104publ191.pdf</ext-link></comment></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="report"><article-title>Regulation (EU) 2016/679 of the European Parliament and of the Council of 27 April 2016 on the protection of natural persons with regard to the processing of personal data and on the free movement of such data, and repealing directive 95/46/EC (General Data Protection Regulation) (text with EEA relevance)</article-title><year>2016</year><access-date>2026-07-28</access-date><publisher-name>Publications Office of the European Union</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://eur-lex.europa.eu/eli/reg/2016/679/oj/eng">https://eur-lex.europa.eu/eli/reg/2016/679/oj/eng</ext-link></comment></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Buhr</surname><given-names>CR</given-names> </name><name name-style="western"><surname>Ernst</surname><given-names>BP</given-names> </name><name name-style="western"><surname>Blaikie</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Assessment of decision-making with locally run and web-based large language models versus human board recommendations in otorhinolaryngology, head and neck surgery</article-title><source>Eur Arch Otorhinolaryngol</source><year>2025</year><month>03</month><volume>282</volume><issue>3</issue><fpage>1593</fpage><lpage>1607</lpage><pub-id pub-id-type="doi">10.1007/s00405-024-09153-3</pub-id><pub-id pub-id-type="medline">39792200</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aubreville</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ganz</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ammeling</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Prediction of tumor board procedural recommendations using large language models</article-title><source>Eur Arch Otorhinolaryngol</source><year>2025</year><month>03</month><volume>282</volume><issue>3</issue><fpage>1619</fpage><lpage>1629</lpage><pub-id pub-id-type="doi">10.1007/s00405-024-08947-9</pub-id><pub-id pub-id-type="medline">39266750</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Banyi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>B</given-names> </name><name name-style="western"><surname>Amanian</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bur</surname><given-names>A</given-names> </name><name name-style="western"><surname>Abdalkhani</surname><given-names>A</given-names> </name></person-group><article-title>Applications of natural language processing in otolaryngology: a scoping review</article-title><source>Laryngoscope</source><year>2025</year><month>09</month><volume>135</volume><issue>9</issue><fpage>3049</fpage><lpage>3063</lpage><pub-id pub-id-type="doi">10.1002/lary.32198</pub-id><pub-id pub-id-type="medline">40309961</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shool</surname><given-names>S</given-names> </name><name name-style="western"><surname>Adimi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Saboori Amleshi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Bitaraf</surname><given-names>E</given-names> </name><name name-style="western"><surname>Golpira</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tara</surname><given-names>M</given-names> </name></person-group><article-title>A systematic review of large language model (LLM) evaluations in clinical medicine</article-title><source>BMC Med Inform Decis Mak</source><year>2025</year><month>03</month><day>7</day><volume>25</volume><issue>1</issue><fpage>117</fpage><pub-id pub-id-type="doi">10.1186/s12911-025-02954-4</pub-id><pub-id pub-id-type="medline">40055694</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>AlSaad</surname><given-names>R</given-names> </name><name name-style="western"><surname>Abd-Alrazaq</surname><given-names>A</given-names> </name><name name-style="western"><surname>Boughorbel</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Multimodal large language models in health care: applications, challenges, and future outlook</article-title><source>J Med Internet Res</source><year>2024</year><month>09</month><day>25</day><volume>26</volume><fpage>e59505</fpage><pub-id pub-id-type="doi">10.2196/59505</pub-id><pub-id pub-id-type="medline">39321458</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hager</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jungmann</surname><given-names>F</given-names> </name><name name-style="western"><surname>Holland</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Evaluation and mitigation of the limitations of large language models in clinical decision-making</article-title><source>Nat Med</source><year>2024</year><month>09</month><volume>30</volume><issue>9</issue><fpage>2613</fpage><lpage>2622</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03097-1</pub-id><pub-id pub-id-type="medline">38965432</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="web"><source>DeepL [Website in German]</source><access-date>2025-05-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.deepl.com/de">https://www.deepl.com/de</ext-link></comment></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="web"><article-title>Calculate confidence limits for a sample proportion</article-title><source>Epitools</source><year>2026</year><access-date>2026-05-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://epitools.ausvet.com.au/ciproportion">https://epitools.ausvet.com.au/ciproportion</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Guan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name></person-group><article-title>Towards trustworthy LLMs: a review on debiasing and dehallucinating in large language models</article-title><source>Artif Intell Rev</source><year>2024</year><volume>57</volume><issue>9</issue><pub-id pub-id-type="doi">10.1007/s10462-024-10896-y</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hofmann</surname><given-names>B</given-names> </name></person-group><article-title>Biases in AI: acknowledging and addressing the inevitable ethical issues</article-title><source>Front Digit Health</source><year>2025</year><volume>7</volume><fpage>1614105</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2025.1614105</pub-id><pub-id pub-id-type="medline">40909204</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>R</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name></person-group><article-title>Effect of delayed treatment on survival of patients with head and neck squamous cell cancer</article-title><source>Sci Rep</source><year>2025</year><month>05</month><day>26</day><volume>15</volume><issue>1</issue><fpage>18366</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-03500-y</pub-id><pub-id pub-id-type="medline">40419633</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>M</given-names> </name><name name-style="western"><surname>Thio</surname><given-names>S</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>A</given-names> </name><name name-style="western"><surname>Siju</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mukit</surname><given-names>W</given-names> </name><name name-style="western"><surname>Kuruvilla</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Grounding large language models in clinical evidence: a retrieval-augmented generation system for querying UK NICE clinical guidelines</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 3, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2510.02967</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>P</given-names> </name><name name-style="western"><surname>Perez</surname><given-names>E</given-names> </name><name name-style="western"><surname>Piktus</surname><given-names>A</given-names> </name><name name-style="western"><surname>Petroni</surname><given-names>F</given-names> </name><name name-style="western"><surname>Karpukhin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Goyal</surname><given-names>N</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Larochelle</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ranzato</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hadsell</surname><given-names>R</given-names> </name><name name-style="western"><surname>Balcan</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>H</given-names> </name></person-group><article-title>Retrieval-augmented generation for knowledge-intensive NLP tasks</article-title><source>NIPS &#x2019;20: Proceedings of the 34th International Conference on Neural Information Processing Systems</source><year>2020</year><access-date>2026-07-28</access-date><publisher-name>Curran Associates Inc</publisher-name><fpage>9459</fpage><lpage>9474</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/abs/10.5555/3495724.3496517">https://dl.acm.org/doi/abs/10.5555/3495724.3496517</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Kraska</surname><given-names>T</given-names> </name><name name-style="western"><surname>Khattab</surname><given-names>O</given-names> </name></person-group><article-title>Recursive language models</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 31, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2512.24601</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ishida</surname><given-names>K</given-names> </name><name name-style="western"><surname>Murakami</surname><given-names>R</given-names> </name><name name-style="western"><surname>Yamanoi</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Real-world application of large language models for automated TNM staging using unstructured gynecologic oncology reports</article-title><source>NPJ Precis Oncol</source><year>2025</year><month>11</month><day>19</day><volume>9</volume><issue>1</issue><fpage>366</fpage><pub-id pub-id-type="doi">10.1038/s41698-025-01157-4</pub-id><pub-id pub-id-type="medline">41261201</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Floudas</surname><given-names>CS</given-names> </name><etal/></person-group><article-title>Matching patients to clinical trials with large language models</article-title><source>Nat Commun</source><year>2024</year><month>11</month><day>18</day><volume>15</volume><issue>1</issue><fpage>9074</fpage><pub-id pub-id-type="doi">10.1038/s41467-024-53081-z</pub-id><pub-id pub-id-type="medline">39557832</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gupta</surname><given-names>S</given-names> </name><name name-style="western"><surname>Basu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nievas</surname><given-names>M</given-names> </name><etal/></person-group><article-title>PRISM: patient records interpretation for semantic clinical trial matching system using large language models</article-title><source>NPJ Digit Med</source><year>2024</year><month>10</month><day>28</day><volume>7</volume><issue>1</issue><fpage>305</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01274-7</pub-id><pub-id pub-id-type="medline">39468259</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rybinski</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kusa</surname><given-names>W</given-names> </name><name name-style="western"><surname>Karimi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hanbury</surname><given-names>A</given-names> </name></person-group><article-title>Learning to match patients to clinical trials using large language models</article-title><source>J Biomed Inform</source><year>2024</year><month>11</month><volume>159</volume><fpage>104734</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2024.104734</pub-id><pub-id pub-id-type="medline">39389283</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Prompt (translated to English with DeepL.com).</p><media xlink:href="ai_v5i1e95081_app1.docx" xlink:title="DOCX File, 4179 KB"/></supplementary-material></app-group></back></article>