<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e95565</article-id><article-id pub-id-type="doi">10.2196/95565</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Evidence Use and Identifier-Conditioned Prior Knowledge in Large Language Model Classification of Oncology Trials Assessed Through Progressive Content Removal and Counterfactual Testing: Comparative Analysis</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Windisch</surname><given-names>Paul</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Koechli</surname><given-names>Carole</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Dennst&#x00E4;dt</surname><given-names>Fabio</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Aebersold</surname><given-names>Daniel M</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zwahlen</surname><given-names>Daniel R</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>F&#x00F6;rster</surname><given-names>Robert</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Schr&#x00F6;der</surname><given-names>Christina</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Radiation Oncology, Kantonsspital Winterthur</institution><addr-line>Brauerstrasse 15</addr-line><addr-line>Winterthur</addr-line><addr-line>Zurich</addr-line><country>Switzerland</country></aff><aff id="aff2"><institution>Department of Radiation Oncology, Inselspital, Bern University Hospital, University of Bern</institution><addr-line>Bern</addr-line><country>Switzerland</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Yun</surname><given-names>Hye Sun</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Clusmann</surname><given-names>Jan</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Paul Windisch, MD, Department of Radiation Oncology, Kantonsspital Winterthur, Brauerstrasse 15, Winterthur, Zurich, 8401, Switzerland, 41 052 266 26 53; <email>paul.windisch@ksw.ch</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>29</day><month>7</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e95565</elocation-id><history><date date-type="received"><day>17</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>03</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>03</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Paul Windisch, Carole Koechli, Fabio Dennst&#x00E4;dt, Daniel M Aebersold, Daniel R Zwahlen, Robert F&#x00F6;rster, Christina Schr&#x00F6;der. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 29.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e95565"/><abstract><sec><title>Background</title><p>Large language models (LLMs) can accurately classify biomedical documents, but strong benchmark performance does not establish that predictions are grounded in the supplied text. In biomedical literature tasks, titles, abstracts, digital object identifiers (DOIs), journal metadata, and trial identifiers may have been seen during pretraining and can trigger parametric knowledge or learned associations.</p></sec><sec><title>Objective</title><p>This study aimed to test whether oncology randomized trial success classification is driven by abstract evidence or by identifier-conditioned prior knowledge, and assess whether models follow counterfactual outcome evidence when it conflicts with original trial identifiers.</p></sec><sec sec-type="methods"><title>Methods</title><p>We evaluated 250 two-arm oncology randomized controlled trials from 7 major journals published between 2005 and 2023, each with a single primary endpoint and previously adjudicated positive or negative ground-truth label. The corpus included 58.4% (146/250) positive and 41.6% (104/250) negative trials. GPT-5.2, Gemini 3 Flash, and Claude Opus 4.5 were queried via vendor APIs under default settings using a single-token output instruction. For each trial, we created 5 deterministic input conditions: title+abstract, title only, DOI only, counterfactual title+abstract in which the primary endpoint outcome statement was minimally flipped, and the same counterfactual input paired with the original DOI to create an identifier-text conflict. Performance was assessed using valid format rate, accuracy, sensitivity, specificity, and <italic>F</italic><sub>1</sub>-score.</p></sec><sec sec-type="results"><title>Results</title><p>The models showed high format adherence, with valid prediction rates of 97.2% to 100%. In the title+abstract condition, all models achieved high and balanced performance (accuracy and <italic>F</italic><sub>1</sub>-score=0.96-0.97; sensitivity=0.96-0.97; specificity=0.96-0.98). Removing evidence reduced performance stepwise: title-only accuracy and <italic>F</italic><sub>1</sub>-score fell to 0.79 to 0.88, and DOI-only performance fell to 0.63-0.67, exceeding the 58.4% majority class baseline but indicating limited identifier-driven signal. Counterfactual edits were concentrated in outcome-bearing text, with the Results and Conclusions sections modified for all trials, whereas the titles and Methods sections required edits in only 5.2% (13/250) and 1.6% (4/250) of trials. Against inverted labels, models followed counterfactual evidence with near-ceiling performance (accuracy and <italic>F</italic><sub>1</sub>-score=0.96-0.99). Reintroducing the original DOI caused little change for GPT-5.2 (accuracy and <italic>F</italic><sub>1</sub>-score=0.99) but modestly reduced <italic>F</italic><sub>1</sub>-scores for Gemini (0.97) and Claude (0.95), mainly through lower sensitivity.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>The evaluated LLMs robustly followed explicit end point statements in abstracts, including when those statements contradicted original trial outcomes. However, above-chance title-only and DOI-only performance, together with small decrements under counterfactual DOI conflicts, showed that identifiers can carry predictive signal and occasionally compete with textual evidence. Progressive content removal combined with counterfactual identifier-text conflicts offers a practical, reproducible audit for grounding in biomedical LLM evaluations.</p></sec></abstract><kwd-group><kwd>natural language processing</kwd><kwd>large language models</kwd><kwd>memory</kwd><kwd>reasoning models</kwd><kwd>context grounding</kwd><kwd>parametric knowledge</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Large language models (LLMs) are increasingly used for biomedical text processing, including automated structuring of electronic health records, clinical trial matching, and literature screening in evidence synthesis workflows [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>].</p><p>However, LLM success on biomedical text tasks is not by itself evidence that the output is grounded in the provided input. A defining feature of modern LLMs is that they store substantial &#x201C;parametric knowledge&#x201D; acquired during pretraining, and in some settings, they can behave like implicit knowledge bases that retrieve facts from model weights rather than from the prompt context [<xref ref-type="bibr" rid="ref4">4</xref>]. In medicine specifically, LLMs have been shown to encode clinically relevant knowledge and perform strongly on medical question answering benchmarks, reinforcing that model parameters can serve as a powerful source of prior information independent of any supplied document [<xref ref-type="bibr" rid="ref5">5</xref>]. For biomedical publishing and evidence synthesis, this distinction is consequential: users may expect the system to follow the text in front of it, whereas the model may instead draw on associations encoded during training with the underlying paper, trial name, venue, or related metadata.</p><p>Concerns about noncontextual sources of task performance are amplified by 2 related phenomena: memorization and data contamination. Previous work has demonstrated that LLMs can leak or reproduce training data under targeted querying, indicating that verbatim or near-verbatim sequences can be memorized and later elicited [<xref ref-type="bibr" rid="ref6">6</xref>]. More recent work has developed formal measures for quantifying extractability and memorization risk, highlighting that the phenomenon can be measured and varies with model and inference choices [<xref ref-type="bibr" rid="ref7">7</xref>]. Separately, systematic analyses have documented that benchmark contamination is widespread in the LLM era and can occur at nontrivial rates, creating the possibility that models appear to &#x201C;solve&#x201D; tasks by recognizing previously seen evaluation items rather than generalizing [<xref ref-type="bibr" rid="ref8">8</xref>]. In biomedical domains, where many inputs (titles, abstracts, and digital object identifiers [DOIs]) are publicly available and plausibly included in pretraining corpora, a model might correctly predict a randomized controlled trial&#x2019;s (RCT) outcome without relying primarily on the outcome evidence contained in the abstract.</p><p>A growing methodological literature has therefore begun to explicitly separate contextual evidence use from prior (parametric) knowledge. Work on knowledge conflicts shows that language models do not integrate context and prior knowledge uniformly and may privilege prior knowledge when the model is &#x201C;familiar&#x201D; with the entity or topic referenced [<xref ref-type="bibr" rid="ref9">9</xref>]. Methods such as contrastive decoding have been proposed to reduce overreliance on encoded priors and improve the use of contextual information when the prompt provides relevant evidence [<xref ref-type="bibr" rid="ref10">10</xref>]. Relatedly, counterfactual paradigms have been used to disentangle parametric from contextual knowledge by constructing prompts in which the provided context is intentionally altered, enabling direct tests of whether a model follows the prompt or defaults to internal memory [<xref ref-type="bibr" rid="ref11">11</xref>]. In the biomedical setting, dedicated benchmarks have further emphasized that knowledge conflicts and unfaithful grounding are practical risks when users provide incomplete, contradictory, or incorrect context [<xref ref-type="bibr" rid="ref12">12</xref>]. Despite these advances, biomedical evaluations that use real documents rarely operationalize a direct, auditably reproducible test of context-grounded evidence use vs identifier-conditioned signal that isolates the informational contribution of the input text from that of identifiers.</p><p>Recent clinical LLM studies similarly caution that apparent medical performance may reflect brittle pattern use or unfaithful grounding rather than reliable use of case-specific evidence. Bedi et al [<xref ref-type="bibr" rid="ref13">13</xref>] showed that medical reasoning benchmark performance can change when standard answer structures are perturbed, supporting the need to distinguish reasoning from recognition. Omar et al [<xref ref-type="bibr" rid="ref14">14</xref>] demonstrated that clinical decision support outputs are vulnerable to adversarial hallucination attacks, with models often elaborating fabricated prompt details despite mitigation attempts. Together with broader calls for task-specific validation before clinical adoption [<xref ref-type="bibr" rid="ref15">15</xref>], these findings motivate evaluations that explicitly test whether models follow supplied biomedical evidence when it conflicts with prior associations.</p><p>This study focused on that methodological gap using the concrete task of classifying oncology RCTs as positive vs negative with respect to primary end point attainment.</p><p>To distinguish context-grounded task performance from identifier-driven prior signal, we leveraged an evaluation framework based on progressive content removal and counterfactual results. The key intuition was diagnostic: if performance remained meaningfully above chance when only identifiers were provided, this would indicate that identifiers carry a predictive signal for the model, potentially through memorized associations, learned metadata correlations, or recognition of trial-specific cues. If predictions then failed to flip under counterfactual results text, then the model&#x2019;s prediction pattern would be more consistent with identifier-conditioned prior signal than with reliance on the supplied outcome evidence.</p><p>We evaluated 3 widely used commercial models under default settings to reflect typical end user deployment conditions. By combining progressive ablations with counterfactual conflict tests, this study aimed to provide an empirically grounded answer to a practical question that is often implicit in biomedical LLM benchmarking: when an LLM classifies a clinical trial abstract correctly, how much of that performance is attributable to the supplied text vs identifier-conditioned prior information?</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><p>This study evaluated whether LLM performance on biomedical trial success classification is primarily driven by information contained in the provided input text or by recognition, parametric knowledge, and learned associations with identifiers.</p><sec id="s2-1"><title>Data and Annotation</title><p>We used an existing dataset that was used in a previous paper from our group consisting of 250 RCTs drawn from 7 major medical journals (<italic>British Medical Journal</italic>, <italic>JAMA</italic>, <italic>JAMA Oncology</italic>, <italic>Journal of Clinical Oncology</italic>, <italic>The Lancet</italic>, <italic>The Lancet Oncology</italic>, and <italic>New England Journal of Medicine</italic>) published between 2005 and 2023 [<xref ref-type="bibr" rid="ref16">16</xref>]. In the aforementioned publications, eligible trials were restricted to designs with exactly 2 arms and a single primary end point, and abstracts were retrieved via PubMed and parsed from text to create the study corpus. For the present analysis, we reused the released dataset and corresponding ground-truth labels.</p><p>Ground-truth labels were adopted from the original dual-annotation procedure performed by 2 authors (PW and CK), in which trials were classified as positive if the primary end point was met and as negative otherwise. The annotation workflow included an initial calibration phase followed by independent labeling and consensus resolution of discrepancies. Full texts or protocols were consulted only when the abstract did not clearly report the primary end point and its results. We did not modify labels or repeat manual annotation for the present analysis.</p></sec><sec id="s2-2"><title>Experimental Conditions</title><p>To distinguish evidence use from identifier-driven recognition, we generated 5 input variants for each RCT. The baseline condition contained the trial title and abstract (title+abstract). A title-only condition contained only the trial title. A DOI-only condition contained only the DOI string. A counterfactual &#x201C;fake results&#x201D; condition contained the potentially edited title and an edited abstract in which the reported primary end point outcome was flipped (positive to negative or negative to positive) while keeping all other content maximally unchanged. Counterfactual abstracts were created by identifying the sentences reporting primary end point attainment (typically in the Results and/or Conclusions sections) and minimally editing the outcome language (eg, reversing whether the primary end point was met or reversing the direction of statistical significance statements tied to the primary end point). Background, design, population, interventions, and secondary outcome text were left unchanged unless modification was required to maintain internal coherence. The original and counterfactual texts are provided in the GitHub repository that is referenced in the Data Availability section. Finally, we input the counterfactual title and abstract together with the original DOI to create a direct conflict between the supplied counterfactual evidence and identifier-conditioned prior information.</p><p>The task for the LLMs was to determine, from the provided information, whether the trial met its primary end point. While this is essentially impossible from a DOI with the exception of a slight bias of high-impact journals to publish positive trials, titles sometimes report results and usually those related to the primary end point. Abstracts usually but not always state which end point was the primary one and explicitly provide the results per end point. To align with the previously published workflow and minimize ambiguity, we used an explicit instruction format requiring the model to output exactly 1 token-level label: &#x201C;POSITIVE&#x201D; or &#x201C;NEGATIVE,&#x201D; all capitalized.</p></sec><sec id="s2-3"><title>Models and Prompts</title><p>The following system prompt was used: &#x201C;You will be provided with information about a randomized controlled oncology trial. Your task will be to classify if the trial was positive, i.e. if it met its primary endpoint, or negative, i.e. if it did not meet its primary endpoint. Your response should be either the word POSITIVE (in all caps) or NEGATIVE (in all caps). Do not output anything else.&#x201D; The user prompt consisted of the respective input variant (title+abstract, title only, DOI only, counterfactual title+abstract, and DOI and counterfactual title+abstract).</p><p>We did not use few-shot examples, retrieval augmentation, or chain-of-thought prompting as the aim was to evaluate default single-label behavior rather than prompt-engineered mitigation strategies.</p><p>Three commercial LLMs were evaluated via their vendor APIs in a local pipeline (Claude [Anthropic], Google Gemini, and GPT-5.2 [OpenAI]). The evaluated model IDs were gpt-5.2-2025-12-11, gemini-3-flash-preview, and claude-opus-4-5-20251101. Models were queried under default vendor settings to reflect typical end user deployment conditions. No additional decoding parameters were set, and no seed was specified. No vendor-specific reasoning control was requested.</p><p>Responses were considered valid only if the returned text matched exactly the &#x201C;POSITIVE&#x201D; or &#x201C;NEGATIVE&#x201D; label after trimming white space. Any other output was recorded as invalid for downstream summaries.</p></sec><sec id="s2-4"><title>Statistical Analysis</title><p>The primary objective was to quantify how model behavior changed as informational content was removed and when counterfactual results created a direct conflict between the provided abstract and the original trial outcome potentially associated with the identifier. Performance for the baseline, title-only, and DOI-only conditions was summarized using <italic>F</italic><sub>1</sub>-score (and accuracy) against the original ground-truth label. For the counterfactual condition, the expected label was defined as the inverse of the original ground truth, and performance was evaluated against this counterfactual target. In addition to per-condition performance, we quantified counterfactual sensitivity by calculating the proportion of trials for which the predicted label differed between the baseline and counterfactual inputs (flip rate), and we summarized invalid-format outputs per model and condition. For the calculation of the <italic>F</italic><sub>1</sub>-score, only valid (ie, correctly formatted predictions) were considered. The 95% CIs were estimated using normal approximation intervals. We did not analyze token-level probabilities or calibration because comparable logits were not consistently available across the evaluated vendor APIs.</p></sec><sec id="s2-5"><title>Ethical Considerations</title><p>This study used publicly available abstracts from published clinical trials. It did not involve human beings, material of human origin, health-related personal data, deceased persons, or human embryos. Therefore, per the applicable regulation (ie, the Swiss Human Research Act and its ordinances), ethics approval was not required [<xref ref-type="bibr" rid="ref17">17</xref>].</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p>Across conditions, the models adhered closely to the required single-token response format (<xref ref-type="table" rid="table1">Table 1</xref>). Valid prediction rates ranged from 97.2% to 100% depending on model and condition (<xref ref-type="fig" rid="figure1">Figure 1</xref>). GPT-5.2 returned 100% valid predictions across all but one condition. Minor format deviations were observed mainly for Gemini (eg, 97.2% valid in the DOI-only condition and 97.6% valid in the counterfactual setting), whereas Claude remained near ceiling (99.2%-100%) across all conditions.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Performance of GPT-5.2, Gemini 3 Flash, and Claude Opus 4.5 under different conditions.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Condition and model</td><td align="left" valign="bottom">Valid predictions (%)</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Sensitivity (95% CI)</td><td align="left" valign="bottom">Specificity (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">Baseline</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5.2</td><td align="left" valign="top">100</td><td align="char" char="." valign="top">0.96 (0.94-0.99)</td><td align="char" char="." valign="top">0.97 (0.94-1.00)</td><td align="char" char="." valign="top">0.96 (0.92-1.00)</td><td align="char" char="." valign="top">0.96 (0.94-0.99)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 3 Flash</td><td align="left" valign="top">99.2</td><td align="char" char="." valign="top">0.96 (0.94-0.99)</td><td align="char" char="." valign="top">0.96 (0.93-0.99)</td><td align="char" char="." valign="top">0.97 (0.94-1.00)</td><td align="char" char="." valign="top">0.96 (0.94-0.99)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Claude Opus 4.5</td><td align="left" valign="top">100</td><td align="char" char="." valign="top">0.97 (0.95-0.99)</td><td align="char" char="." valign="top">0.97 (0.94-1.00)</td><td align="char" char="." valign="top">0.98 (0.95-1.00)</td><td align="char" char="." valign="top">0.97 (0.95-0.99)</td></tr><tr><td align="left" valign="top" colspan="6">Title only</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5.2</td><td align="left" valign="top">100</td><td align="char" char="." valign="top">0.82 (0.78-0.87)</td><td align="char" char="." valign="top">0.82 (0.75-0.88)</td><td align="char" char="." valign="top">0.84 (0.77-0.91)</td><td align="char" char="." valign="top">0.82 (0.77-0.87)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 3 Flash</td><td align="left" valign="top">99.2</td><td align="char" char="." valign="top">0.88 (0.84-0.92)</td><td align="char" char="." valign="top">0.88 (0.83-0.93)</td><td align="char" char="." valign="top">0.88 (0.81-0.94)</td><td align="char" char="." valign="top">0.88 (0.84-0.92)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Claude Opus 4.5</td><td align="left" valign="top">100</td><td align="char" char="." valign="top">0.79 (0.74-0.84)</td><td align="char" char="." valign="top">0.79 (0.73-0.86)</td><td align="char" char="." valign="top">0.78 (0.70-0.86)</td><td align="char" char="." valign="top">0.78 (0.73-0.83)</td></tr><tr><td align="left" valign="top" colspan="6">DOI<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> only</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5.2</td><td align="left" valign="top">100</td><td align="char" char="." valign="top">0.67 (0.61-0.73)</td><td align="char" char="." valign="top">0.76 (0.69-0.83)</td><td align="char" char="." valign="top">0.55 (0.45-0.64)</td><td align="char" char="." valign="top">0.66 (0.60-0.71)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 3 Flash</td><td align="left" valign="top">97.2</td><td align="char" char="." valign="top">0.63 (0.57-0.69)</td><td align="char" char="." valign="top">0.55 (0.47-0.63)</td><td align="char" char="." valign="top">0.75 (0.67-0.84)</td><td align="char" char="." valign="top">0.63 (0.57-0.69)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Claude Opus 4.5</td><td align="left" valign="top">99.6</td><td align="char" char="." valign="top">0.63 (0.57-0.69)</td><td align="char" char="." valign="top">0.63 (0.55-0.71)</td><td align="char" char="." valign="top">0.63 (0.54-0.73)</td><td align="char" char="." valign="top">0.63 (0.57-0.69)</td></tr><tr><td align="left" valign="top" colspan="6">Counterfactual</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5.2</td><td align="left" valign="top">99.2</td><td align="char" char="." valign="top">0.99 (0.98-1.00)</td><td align="char" char="." valign="top">0.99 (0.97-1.00)</td><td align="char" char="." valign="top">0.99 (0.98-1.00)</td><td align="char" char="." valign="top">0.99 (0.98-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 3 Flash</td><td align="left" valign="top">97.6</td><td align="char" char="." valign="top">0.98 (0.96-1.00)</td><td align="char" char="." valign="top">0.96 (0.92-1.00)</td><td align="char" char="." valign="top">0.99 (0.98-1.00)</td><td align="char" char="." valign="top">0.98 (0.96-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Claude Opus 4.5</td><td align="left" valign="top">99.2</td><td align="char" char="." valign="top">0.96 (0.94-0.99)</td><td align="char" char="." valign="top">0.92 (0.87-0.97)</td><td align="char" char="." valign="top">0.99 (0.98-1.00)</td><td align="char" char="." valign="top">0.96 (0.94-0.99)</td></tr><tr><td align="left" valign="top" colspan="6">Counterfactual+DOI</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5.2</td><td align="left" valign="top">100</td><td align="char" char="." valign="top">0.99 (0.97-1.00)</td><td align="char" char="." valign="top">0.99 (0.97-1.00)</td><td align="char" char="." valign="top">0.99 (0.97-1.00)</td><td align="char" char="." valign="top">0.99 (0.97-1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 3 Flash</td><td align="left" valign="top">97.6</td><td align="char" char="." valign="top">0.97 (0.94-0.99)</td><td align="char" char="." valign="top">0.92 (0.87-0.97)</td><td align="char" char="." valign="top">1.00 (1.00-1.00)</td><td align="char" char="." valign="top">0.97 (0.94-0.99)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Claude Opus 4.5</td><td align="left" valign="top">99.2</td><td align="char" char="." valign="top">0.95 (0.92-0.98)</td><td align="char" char="." valign="top">0.90 (0.84-0.96)</td><td align="char" char="." valign="top">0.98 (0.96-1.00)</td><td align="char" char="." valign="top">0.95 (0.92-0.97)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>DOI: digital object identifier.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Valid prediction rates of GPT-5.2, Gemini 3 Flash, and Claude Opus 4.5 under different conditions. DOI: digital object identifier.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95565_fig01.png"/></fig><p>The confusion matrices for all models and conditions are provided in <xref ref-type="fig" rid="figure2">Figure 2</xref>. The corpus of 250 trials contained 146 (58.4%) positive trials and 104 (41.6%) negative trials; always predicting the majority class would yield a 58.4% accuracy and a macro&#x2013;<italic>F</italic><sub>1</sub>-score of 0.37. When provided with the full title and abstract, all 3 models achieved high and tightly clustered classification performance (accuracy=0.96-0.97; <italic>F</italic><sub>1</sub>-score=0.96-0.97; <xref ref-type="table" rid="table1">Table 1</xref>). Sensitivity and specificity were similarly high (sensitivity=0.96-0.97; specificity=0.96-0.98), indicating balanced performance for positive and negative trials when the complete abstract text was available.</p><p>Removing abstract content reduced performance in a stepwise manner. Under the title-only condition, accuracy fell to 0.79 to 0.88, and the <italic>F</italic><sub>1</sub>-score fell to 0.78 to 0.88, with Gemini 3 Flash performing best (accuracy and <italic>F</italic><sub>1</sub>-score=0.88) and Claude Opus 4.5 performing worst (accuracy=0.79; <italic>F</italic><sub>1</sub>-score=0.78). Sensitivity and specificity remained broadly comparable within each model (GPT-5.2=0.82 vs 0.84; Gemini=0.88 vs 0.88; Claude=0.79 vs 0.78, respectively), suggesting that performance degradation with titles alone was not driven solely by one-sided misclassification. With DOI-only inputs, performance decreased further for all models (accuracy=0.63-0.67; <italic>F</italic><sub>1</sub>-score=0.63-0.66). Error profiles diverged by model: GPT-5.2 showed higher sensitivity than specificity (0.76 vs 0.55), whereas Gemini showed the reverse pattern (0.55 vs 0.75), and Claude remained relatively symmetric (0.63 vs 0.63). The <italic>F</italic><sub>1</sub>-scores for all models and conditions are visualized in <xref ref-type="fig" rid="figure3">Figure 3</xref>.</p><p>To generate counterfactual abstracts, edits were concentrated in outcome-bearing portions of the text. Across 100% (250/250) of the trials, both the Results and Conclusions sections required modification. In contrast, edits were rarely required in the title (13/250, 5.2%) or Methods section (4/250, 1.6%), and the Introduction section did not require modification (0%).</p><p>Against the inverted ground-truth labels, the models followed the counterfactual evidence with near-ceiling performance. In the counterfactual condition, accuracy and <italic>F</italic><sub>1</sub>-score ranged from 0.96 to 0.99 (GPT-5.2=0.99; Gemini=0.98; Claude=0.96). Specificity remained high (0.99 for all 3 models), whereas Claude&#x2019;s decrease relative to GPT-5.2 and Gemini was driven mainly by lower sensitivity (0.92, 95% CI 0.87-0.97).</p><p>When the original identifier was reintroduced (counterfactual+DOI), performance changed minimally for GPT-5.2 (accuracy and <italic>F</italic><sub>1</sub>-score=0.99) but decreased modestly for Gemini (accuracy and <italic>F</italic><sub>1</sub>-score=0.97) and Claude (accuracy and <italic>F</italic><sub>1</sub>-score=0.95). Notably, Gemini maintained perfect specificity (1.00) in this condition (95% CI 1.00-1.00) alongside reduced sensitivity (0.92), whereas Claude showed a larger sensitivity drop (0.90) with specificity still high (0.98).</p><p>Information on additional input conditions (abstract only, title+DOI, and counterfactual title only), as well as examples of counterfactual titles and abstracts, can be found in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Confusion matrices of GPT-5.2, Gemini 3 Flash, and Claude Opus 4.5 under different conditions. DOI: digital object identifier.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95565_fig02.png"/></fig><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Heat map of the <italic>F</italic><sub>1</sub>-scores of GPT-5.2, Gemini 3 Flash, and Claude Opus 4.5 under different conditions. The numbers in parentheses indicate the 95% CIs. DOI: digital object identifier.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95565_fig03.png"/></fig></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>All 3 commercial models achieved high performance when given full title and abstract information (accuracy and <italic>F</italic><sub>1</sub>-score clustered around 0.96-0.97), with only minor rates of invalid-format outputs. Performance decreased stepwise as informational content was removed: with title-only inputs, accuracy dropped to a range of 0.79 to 0.88, and with DOI-only inputs, it fell further to a range of 0.63 to 0.67. In contrast, when the abstract&#x2019;s primary outcome statement was counterfactually flipped, the models followed the altered evidence and achieved near-ceiling performance against the inverted labels (accuracy and <italic>F</italic><sub>1</sub>-score=0.96-0.99). Reintroducing the real DOI alongside the counterfactual abstract minimally affected GPT-5.2 but led to modest performance reductions for Gemini and Claude, primarily via reduced sensitivity, suggesting occasional identifier-driven interference when DOI and results text conflicted.</p></sec><sec id="s4-2"><title>Interpretation and Comparison to Prior Work</title><p>Interpreting these results in the context of prior work, the strong baseline performance is consistent with broader evidence that LLMs can perform competitively on biomedical language tasks [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. However, high task performance alone does not establish that the prediction is grounded in the provided document rather than influenced by parametric knowledge or learned associations, an issue long discussed in the framing of language models as implicit knowledge bases [<xref ref-type="bibr" rid="ref4">4</xref>]. The stepwise ablation results speak directly to that distinction. Title-only performance well above chance can plausibly arise from a mixture of (1) genuine, weakly informative cues in titles (eg, disease setting, intervention class, and end point hints); and (2) recognition of well-known trials from pretraining exposure. DOI-only inputs are informative because the DOI string contains minimal semantic content for end point attainment. Above-chance performance under DOI-only conditions is therefore compatible with identifier-triggered recall, learned metadata correlations, or other DOI-associated regularities.</p><p>However, the counterfactual results indicate that the models&#x2019; predictions were usually dominated by the supplied counterfactual outcome evidence rather than by any identifier-associated prior signal. When the outcome-bearing sentences in the abstract were minimally edited to flip the primary end point conclusion, the models overwhelmingly produced the inverted label, indicating sensitivity to the supplied text even when that text contradicted the original outcome potentially associated with the trial identifier. At the same time, the small but consistent decrement for some models when the real DOI was appended to the counterfactual abstract suggests that identifiers can still exert a measurable pull on the model&#x2019;s decision, echoing findings that models do not integrate context and prior knowledge uniformly and can privilege priors when they are &#x201C;familiar&#x201D; with an entity [<xref ref-type="bibr" rid="ref9">9</xref>]. It is important to emphasize that our counterfactual flips created direct, high-salience evidence conflicts. In more realistic settings, users may supply incomplete or subtly contradictory evidence, where LLMs can struggle with conflict resolution in biomedical contexts [<xref ref-type="bibr" rid="ref12">12</xref>]. Taken together, the findings support a nuanced conclusion: the models can robustly follow explicit outcome statements in abstracts (strong context-grounded behavior under explicit outcome evidence) yet still display some degree of susceptibility to identifier-driven priors when content is sparse or when identifiers are introduced.</p></sec><sec id="s4-3"><title>Strengths and Limitations</title><p>Several strengths support the interpretability and practical relevance of these findings. First, the design operationalizes the contrast between context-grounded evidence use and identifier-conditioned prior signal using deterministic, logged perturbations (progressive content removal plus counterfactual outcome edits), which makes the diagnostic logic transparent and reproducible. Second, the evaluation used a corpus of real oncology RCT abstracts spanning multiple high-impact journals and many publication years, aligning the test distribution with how LLMs are used in evidence synthesis and clinical research workflows rather than relying solely on synthetic benchmarks. Third, the single-token output constraint (&#x201C;POSITIVE&#x201D; and &#x201C;NEGATIVE&#x201D;) reduced ambiguity in scoring and minimized the confounding role of verbose explanations, whereas the high valid output rates indicate that the API-based setup was stable across conditions. Fourth, comparing multiple commercial models under default settings improved external validity for typical end user deployment, where practitioners rarely tune decoding parameters or implement specialized grounding interventions.</p><p>Several limitations should temper overgeneralization. First, this study focused on a single binary task (primary end point met vs not met) within oncology RCTs. The balance between context-grounded evidence use and identifier-conditioned prior signal may differ for tasks requiring finer-grained extraction, multi-label judgments, or synthesis across multiple documents. Second, DOI-only above-chance performance could partly reflect indirect correlations encoded in DOI structure (publisher prefix, journal family, and year) rather than memorized trial-specific outcome associations. Additional controls (eg, DOI shuffling across papers and synthetic DOIs matched on prefix and year) would help isolate the mechanism. Third, counterfactual editing was designed to be minimal, but any manual or rule-based editing procedure risks introducing artifacts (lexical cues, unnatural phrasing, or consistency breaks) that models could exploit. Even though edits were concentrated where outcomes were stated, future work could quantify the detectability of edits or use blinded human review to ensure counterfactual naturalness. Fourth, the models were queried via vendor APIs with default settings and without a fixed seed. While this reflects realistic use, it limits ANOVA due to decoding stochasticity and complicates strict reproducibility across future model snapshots. Finally, we did not evaluate model calibration, abstention behavior, or uncertainty reporting&#x2014;properties that may be crucial when deploying such systems in safety-critical evidence workflows.</p></sec><sec id="s4-4"><title>Outlook</title><p>Future research could try to strengthen causal attribution of identifier effects by adding negative controls: swapping DOIs between trials, adding semantically irrelevant identifiers, or using &#x201C;matched&#x201D; synthetic DOIs, which would help separate true memorized mapping from metadata correlations. A second direction is to broaden task coverage: applying the same progressive ablation and counterfactual conflict framework to other biomedical natural language processing tasks (population, intervention, comparator, and outcome extraction; effect direction and magnitude; toxicity end points; and comparative effectiveness statements) would test whether strong counterfactual sensitivity generalizes beyond this relatively explicit classification problem. A third direction is temporal and contamination-robust evaluation: constructing time-split test sets consisting of trials published after known model training cutoffs (or using newly published or embargoed material where feasible) would reduce the plausibility of trial-specific parametric associations and better isolate context-grounded task performance.</p></sec><sec id="s4-5"><title>Conclusions</title><p>In conclusion, this study suggests that the evaluated models were highly sensitive to explicit outcome statements in oncology RCT abstracts, as shown by near-ceiling performance on counterfactual abstracts that directly contradicted the original results. At the same time, above-chance performance with titles&#x2014;and especially with DOI-only inputs&#x2014;indicates that identifiers can carry predictive signal for these models, consistent with identifier-associated predictive signal that is at least partly independent of the provided abstract text. The modest degradation in performance observed when reintroducing real DOIs into counterfactual abstracts further implies that identifier-triggered priors can occasionally compete with textual evidence.</p></sec></sec></body><back><ack><p>The authors declare the use of generative AI (GenAI) in the research and writing process. According to the Generative AI Delegation Taxonomy (2025), the following tasks were delegated to GenAI tools under full human supervision: proofreading and editing. The GenAI tool used was GPT-5.5 (OpenAI). Responsibility for the final manuscript lies entirely with the authors. GenAI tools are not listed as authors and do not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>This study was funded by the Swiss Cancer Research foundation (grant KFS-6477-08-2025-R).</p></sec><sec><title>Data Availability</title><p>The datasets generated or analyzed during this study are available in a GitHub repository [<xref ref-type="bibr" rid="ref18">18</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: PW</p><p>Data curation: PW, CK</p><p>Formal analysis: PW</p><p>Methodology: PW, CS</p><p>Project administration: DRZ</p><p>Supervision: DRZ</p><p>Writing&#x2014;original draft: PW, CS</p><p>Writing&#x2014;review and editing: CK, FD, DMA, DRZ, RF</p><p>All authors read and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">DOI</term><def><p>digital object identifier</p></def></def-item><def-item><term id="abb2">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb3">RCT</term><def><p>randomized controlled trial</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Floudas</surname><given-names>CS</given-names> </name><etal/></person-group><article-title>Matching patients to clinical trials with large language models</article-title><source>Nat Commun</source><year>2024</year><month>11</month><day>18</day><volume>15</volume><issue>1</issue><fpage>9074</fpage><pub-id pub-id-type="doi">10.1038/s41467-024-53081-z</pub-id><pub-id pub-id-type="medline">39557832</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guevara</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Large language models to identify social determinants of health in electronic health records</article-title><source>NPJ Digit Med</source><year>2024</year><month>01</month><day>11</day><volume>7</volume><issue>1</issue><fpage>6</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00970-0</pub-id><pub-id pub-id-type="medline">38200151</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Landschaft</surname><given-names>A</given-names> </name><name name-style="western"><surname>Antweiler</surname><given-names>D</given-names> </name><name name-style="western"><surname>Mackay</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Implementation and evaluation of an additional GPT-4-based reviewer in PRISMA-based medical systematic literature reviews</article-title><source>Int J Med Inform</source><year>2024</year><month>09</month><volume>189</volume><fpage>105531</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2024.105531</pub-id><pub-id pub-id-type="medline">38943806</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Petroni</surname><given-names>F</given-names> </name><name name-style="western"><surname>Rockt&#x00E4;schel</surname><given-names>T</given-names> </name><name name-style="western"><surname>Riedel</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Language models as knowledge bases?</article-title><source>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</source><year>2019</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>2463</fpage><lpage>2473</lpage><pub-id pub-id-type="doi">10.18653/v1/D19-1250</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Azizi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language models encode clinical knowledge</article-title><source>Nature</source><year>2023</year><month>08</month><volume>620</volume><issue>7972</issue><fpage>172</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id><pub-id pub-id-type="medline">37438534</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Carlini</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tramer</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wallace</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Extracting training data from large language models</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 14, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.2012.07805</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Hayes</surname><given-names>J</given-names> </name><name name-style="western"><surname>Swanberg</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chaudhari</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Measuring memorization in language models via probabilistic extraction</article-title><source>Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies</source><year>2025</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>9266</fpage><lpage>9291</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.naacl-long.469</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Guerin</surname><given-names>F</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>C</given-names> </name></person-group><article-title>An open-source data contamination report for large language models</article-title><source>Findings of the Association for Computational Linguistics: EMNLP 2024</source><year>2024</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>528</fpage><lpage>541</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.findings-emnlp.30</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Du</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sn&#x00E6;bjarnarson</surname><given-names>V</given-names> </name><name name-style="western"><surname>Stoehr</surname><given-names>N</given-names> </name><name name-style="western"><surname>White</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schein</surname><given-names>A</given-names> </name><name name-style="western"><surname>Cotterell</surname><given-names>R</given-names> </name></person-group><article-title>Context versus prior knowledge in language models</article-title><source>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics</source><year>2024</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>13211</fpage><lpage>13235</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.acl-long.714</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Monti</surname><given-names>E</given-names> </name><name name-style="western"><surname>Lehmann</surname><given-names>J</given-names> </name><name name-style="western"><surname>Assem</surname><given-names>H</given-names> </name></person-group><article-title>Enhancing contextual understanding in large language models through contrastive decoding</article-title><source>Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies</source><year>2024</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>4225</fpage><lpage>4237</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.naacl-long.237</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Neeman</surname><given-names>E</given-names> </name><name name-style="western"><surname>Aharoni</surname><given-names>R</given-names> </name><name name-style="western"><surname>Honovich</surname><given-names>O</given-names> </name><name name-style="western"><surname>Choshen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Szpektor</surname><given-names>I</given-names> </name><name name-style="western"><surname>Abend</surname><given-names>O</given-names> </name></person-group><article-title>DisentQA: disentangling parametric and contextual knowledge with counterfactual question answering</article-title><source>Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>10056</fpage><lpage>10070</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.acl-long.559</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Bornet</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>N</given-names> </name><name name-style="western"><surname>Teodoro</surname><given-names>D</given-names> </name></person-group><article-title>HealthContradict: evaluating biomedical knowledge conflicts in language models</article-title><source>NPJ Digit Med</source><year>2026</year><month>01</month><day>21</day><volume>9</volume><issue>1</issue><fpage>152</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-02336-0</pub-id><pub-id pub-id-type="medline">41565976</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bedi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>P</given-names> </name><name name-style="western"><surname>Koyejo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>N</given-names> </name></person-group><article-title>Fidelity of medical reasoning in large language models</article-title><source>JAMA Netw Open</source><year>2025</year><month>08</month><day>1</day><volume>8</volume><issue>8</issue><fpage>e2526021</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.26021</pub-id><pub-id pub-id-type="medline">40779272</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Omar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sorin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Collins</surname><given-names>JD</given-names> </name><etal/></person-group><article-title>Multi-model assurance analysis showing large language models are highly vulnerable to adversarial hallucination attacks during clinical decision support</article-title><source>Commun Med (Lond)</source><year>2025</year><month>08</month><day>2</day><volume>5</volume><issue>1</issue><fpage>330</fpage><pub-id pub-id-type="doi">10.1038/s43856-025-01021-3</pub-id><pub-id pub-id-type="medline">40753316</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shah</surname><given-names>NH</given-names> </name><name name-style="western"><surname>Entwistle</surname><given-names>D</given-names> </name><name name-style="western"><surname>Pfeffer</surname><given-names>MA</given-names> </name></person-group><article-title>Creation and adoption of large language models in medicine</article-title><source>JAMA</source><year>2023</year><month>09</month><day>5</day><volume>330</volume><issue>9</issue><fpage>866</fpage><lpage>869</lpage><pub-id pub-id-type="doi">10.1001/jama.2023.14217</pub-id><pub-id pub-id-type="medline">37548965</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koechli</surname><given-names>C</given-names> </name><name name-style="western"><surname>Dennst&#x00E4;dt</surname><given-names>F</given-names> </name><name name-style="western"><surname>Schr&#x00F6;der</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Large language models for supporting clear writing and detecting spin in randomized controlled trials in oncology: comparative analysis of GPT models and prompts</article-title><source>JMIR Cancer</source><year>2026</year><month>01</month><day>21</day><volume>12</volume><fpage>e78221</fpage><pub-id pub-id-type="doi">10.2196/78221</pub-id><pub-id pub-id-type="medline">41564336</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="web"><article-title>Regulation of human research in Switzerland</article-title><source>Federal Office of Public Health</source><access-date>2026-07-03</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.bag.admin.ch/en/regulation-of-human-research-in-switzerland">https://www.bag.admin.ch/en/regulation-of-human-research-in-switzerland</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="web"><article-title>windisch-paul/llm_memory</article-title><source>GitHub</source><access-date>2026-07-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/windisch-paul/llm_memory">https://github.com/windisch-paul/llm_memory</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary material containing representative counterfactual trial examples and additional input conditions.</p><media xlink:href="ai_v5i1e95565_app1.docx" xlink:title="DOCX File, 39 KB"/></supplementary-material></app-group></back></article>