<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e100772</article-id><article-id pub-id-type="doi">10.2196/100772</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>A Patient Simulation Framework for Risk Assessment of Conversational Health Care AI: Development and Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Shawon</surname><given-names>Md Tanvir Rouf</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Irbaz</surname><given-names>Mohammad Sabik</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Elyazori</surname><given-names>Hadeel R A</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Resapu</surname><given-names>Keerti Reddy</given-names></name><degrees>MS, BDS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lin</surname><given-names>Yili</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cardenas</surname><given-names>Vladimir Franzuela</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Eklou</surname><given-names>K Pierre</given-names></name><degrees>DNP, PMHNP-BC</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Alemi</surname><given-names>Farrokh</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lybarger</surname><given-names>Kevin</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Computer Science, College of Engineering and Computing, George Mason University</institution><addr-line>Nguyen Engineering Building, 4511 Patriot Cir</addr-line><addr-line>Fairfax</addr-line><addr-line>VA</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Information Sciences and Technology, College of Engineering and Computing, George Mason University</institution><addr-line>Fairfax</addr-line><addr-line>VA</addr-line><country>United States</country></aff><aff id="aff3"><institution>Department of Health Administration and Policy, College of Public Health, George Mason University</institution><addr-line>Fairfax</addr-line><addr-line>VA</addr-line><country>United States</country></aff><aff id="aff4"><institution>School of Nursing, College of Public Health, George Mason University</institution><addr-line>Fairfax</addr-line><addr-line>VA</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Liu</surname><given-names>Hongfang</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Chipps</surname><given-names>Jennifer</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Ajayi</surname><given-names>Rhoda</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Md Tanvir Rouf Shawon, BS, Department of Computer Science, College of Engineering and Computing, George Mason University, Nguyen Engineering Building, 4511 Patriot Cir, Fairfax, VA, 22032, United States, 1 571-663-5183; <email>mshawon@gmu.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>5</day><month>10</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e100772</elocation-id><history><date date-type="received"><day>08</day><month>05</month><year>2026</year></date><date date-type="rev-recd"><day>21</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>08</day><month>09</month><year>2026</year></date></history><copyright-statement>&#x00A9; Md Tanvir Rouf Shawon, Mohammad Sabik Irbaz, Hadeel R A Elyazori, Keerti Reddy Resapu, Yili Lin, Vladimir Franzuela Cardenas, K Pierre Eklou, Farrokh Alemi, Kevin Lybarger. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 5.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e100772"/><abstract><sec><title>Background</title><p>Conversational AI systems are increasingly being deployed in health care for clinical decision support, but their performance varies substantially across patient communication styles, health literacy levels, and behavioral patterns. Static benchmarks cannot capture multiturn dynamics through which this variation compounds, and no current evaluation framework implements structured AI risk management guidance for conversational health care AI. The result is a structural risk: AI systems may perform well in aggregate while failing disproportionately for the populations they are intended to help.</p></sec><sec><title>Objective</title><p>This study aimed to develop and validate a patient simulation framework that aligns with the National Institute of Standards and Technology (NIST) AI Risk Management Framework (AI RMF) Map and Measure functions, providing an empirical basis for identifying and characterizing performance risks in conversational clinical AI across medical, linguistic, and behavioral patient variations. We applied the framework to a conversational decision aid for antidepressant selection in major depressive disorder (the AI decision aid).</p></sec><sec sec-type="methods"><title>Methods</title><p>The simulator integrated three profile dimensions: (1) medical profiles constructed from All of Us electronic health records using risk ratio gating; (2) linguistic profiles modeling a health literacy gradient and condition-specific communication; and (3) behavioral profiles representing cooperative, distracted, and adversarial engagement. We generated 500 simulated conversations and evaluated profile fidelity through human annotation and a large language model (LLM) judge, and then assessed downstream effects on the AI decision aid&#x2019;s concept retrieval and antidepressant recommendations.</p></sec><sec sec-type="results"><title>Results</title><p>The patient simulator expressed medical concepts with high fidelity (96.6% accuracy across 8210 concepts), with substantial human interannotator agreement (&#x03BA;=0.73) and LLM-judge agreement against human annotators (&#x03BA;=0.78). Behavioral profiles were reliably distinguished (&#x03BA;=0.93; near perfect agreement), and linguistic profiles showed substantial agreement at the lower bound of the substantial range (&#x03BA;=0.61), which can be considered adequate to support profile-level analysis. The framework revealed monotonic degradation in AI decision aid performance across the health literacy gradient. Rank-1 concept retrieval increased from 47.6% for limited health literacy to 81.9% for proficient health literacy, with corresponding declines in antidepressant recommendation accuracy.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Patient simulation grounded in the NIST AI RMF exposes measurable performance risks in conversational health care AI that static benchmarks miss, with direct equity implications. Health literacy operates as a structural risk factor, with degraded performance concentrated in patients carrying the greatest burden of psychiatric illness. The framework supports targeted risk-mitigation interventions before deployment. While we evaluated the framework only on antidepressant selection, extending it to other clinical decision-aid tasks remains an assignment for future work.</p></sec></abstract><kwd-group><kwd>natural language generation</kwd><kwd>consumer health</kwd><kwd>health informatics</kwd><kwd>patient simulation</kwd><kwd>conversational agents</kwd><kwd>AI risk management</kwd><kwd>trustworthy AI</kwd><kwd>large language models</kwd><kwd>depression management</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Conversational AI systems are being increasingly deployed in health care to support clinical decision-making, patient engagement, and care accessibility [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. Yet these systems face a fundamental evaluation gap: patients communicate the same clinical information in vastly different ways depending on health literacy, psychological state, and interaction style [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. A patient who says, &#x201C;my morning pill for my nerves,&#x201D; and another who reports, &#x201C;20 mg of fluoxetine for generalized anxiety disorder,&#x201D; may describe identical clinical realities, but a conversational agent that handles the second statement better than the first introduces a structural bias against vulnerable populations. Recent work demonstrates that large language model (LLM) clinical outputs shift measurably when patient messages are perturbed with stylistic, syntactic, or demographic variation, with disparities concentrated in vulnerable subgroups and amplified in conversational settings [<xref ref-type="bibr" rid="ref7">7</xref>]. There is no standardized method to assess whether AI performance remains equitable across this variation [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>], so disparities are likely to surface only after deployment, with disproportionate impact on populations facing existing barriers to equitable care. LLM-based approaches lower barriers to building such systems [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>] but heighten the need for rigorous, scalable risk assessment that can characterize performance across diverse patient communication patterns.</p><p>Static benchmarks cannot capture these risks because they bypass the multiturn dynamics through which communication variation compounds [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. Patient simulation offers a scalable alternative: simulated patients with controlled clinical, linguistic, and behavioral characteristics can systematically probe system performance across populations that would be difficult, slow, or ethically complex to recruit for live evaluation [<xref ref-type="bibr" rid="ref14">14</xref>]. Simulation-based methods are also being used more broadly to quantify LLM risk in health care, for example, to estimate hazard-to-harm probabilities for regulatory risk assessment of LLM-based medical devices [<xref ref-type="bibr" rid="ref15">15</xref>]. What is missing is a structured framework for connecting patient simulation to AI risk governance. The National Institute of Standards and Technology (NIST) AI Risk Management Framework (AI RMF) provides domain-agnostic guidance for identifying, assessing, and managing AI risks [<xref ref-type="bibr" rid="ref16">16</xref>], but no existing patient simulation framework operationalizes this guidance for conversational health care AI [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. This work introduces such a framework. Patient simulator construction supports the Map function by structuring risk identification across medical, linguistic, and behavioral profile dimensions. Simulator-AI interaction supports the Measure function by surfacing performance variation across the structured profile space. The framework surfaces the evidence needed for AI RMF&#x2013;aligned risk analysis. Extending this evidence into deployment-specific risk analyses is left for future work.</p><p>We present a patient simulator grounded in real-world clinical data from the All of Us Research Program [<xref ref-type="bibr" rid="ref17">17</xref>], evaluated on a conversational decision aid for antidepressant selection in major depressive disorder. The simulator integrates three profile dimensions: (1) medical profiles constructed from electronic health record (EHR) data through a risk ratio (RR)&#x2013;based feature selection process that prioritizes outcome-relevant clinical features while preserving statistical independence and clinical coherence; (2) linguistic profiles modeling health literacy variation [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>] and condition-specific communication patterns [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]; and (3) behavioral profiles representing empirically derived interaction patterns, including cooperative, distracted, and adversarial engagement [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. The framework generalizes across health care tasks and conditions. This study evaluates it on antidepressant selection while also addressing the following research questions (RQs):</p><list list-type="bullet"><list-item><p>RQ1: How effectively can structured EHR data be transformed into risk-aware medical profiles aligned with AI trustworthiness principles?</p></list-item><list-item><p>RQ2: How can simulated patients combining medical, linguistic, and behavioral information produce realistic and distinguishable conversational behavior?</p></list-item><list-item><p>RQ3: How does simulated patient variation across medical, linguistic, and behavioral profiles affect the performance of a conversational clinical decision aid, and which patient subgroups are most affected?</p></list-item></list><p>This work provides the following: (1) a patient simulation framework that aligns with the NIST AI RMF Map and Measure functions for conversational health care AI, integrating medical, linguistic, and behavioral profiles; (2) empirical evidence that health literacy creates a monotonic performance gradient in conversational clinical AI, identifying a concrete equity risk; (3) a RR-based algorithm for generating outcome-relevant, auditable medical profiles from EHR data; and (4) a controlled perturbation methodology using ontology-aware semantic substitution for validating both human annotators and LLM judges. The work was conducted by a multidisciplinary team integrating clinical psychiatric and mental health nursing expertise, clinical informatics, and natural language processing, with guidance from a patient and clinician advisory board. The code, source data, and simulated interactions are publicly available [<xref ref-type="bibr" rid="ref24">24</xref>].</p></sec><sec id="s1-2"><title>Related Work</title><sec id="s1-2-1"><title>Overview</title><p>In this section, we review related work on user simulation, patient simulation, and behavioral and linguistic modeling for conversational AI. User simulation provides scalable methods for evaluating conversational agents across domains, and patient simulation adapts these methods for clinical dialogue and diagnostic reasoning.</p></sec><sec id="s1-2-2"><title>User Simulation</title><p>User simulation enables controlled, scalable evaluation of interactive systems without requiring human trials. Foundational surveys span information access, dialogue modeling, and recommendation systems [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref27">27</xref>], while recent work leverages LLMs to generate context-aware user behavior through dual-model architectures combining generators and verifiers [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>] and session-level search simulation [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. In health care, synthetic users with clinical profiles evaluate decision support and coaching systems [<xref ref-type="bibr" rid="ref32">32</xref>], establishing user simulation as a critical method for assessing reliability and safety in AI systems. Mental health conversational AI research has historically split between technical evaluation of response quality and clinical evaluation of patient outcomes, with limited integration across the two [<xref ref-type="bibr" rid="ref33">33</xref>].</p></sec><sec id="s1-2-3"><title>Patient Simulation</title><p>Patient simulation involves the construction of virtual patients that engage in interactive clinical dialogues with human or automated clinicians [<xref ref-type="bibr" rid="ref34">34</xref>]. Frameworks vary widely in patient state representation, from hand-crafted profiles to EHR-grounded models, and in how they control disclosure, tone, and clinical accuracy [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. Prior work has often treated patient simulation as a single unified problem, but it involves two distinct technical challenges: (1) constructing clinically valid patient representations and (2) simulating interactive dialogue that expresses those representations over time [<xref ref-type="bibr" rid="ref36">36</xref>]. Recent systems have integrated LLMs to enhance expressiveness and flexibility [<xref ref-type="bibr" rid="ref37">37</xref>], constructed structured knowledge bases from clinical notes to support medical intake tasks [<xref ref-type="bibr" rid="ref38">38</xref>], and embedded risk-aware feedback mechanisms [<xref ref-type="bibr" rid="ref34">34</xref>].</p></sec><sec id="s1-2-4"><title>Data-Driven Medical Realism</title><p>Early systems used deterministic state machines or probabilistic sequence models to generate medical histories. Synthea [<xref ref-type="bibr" rid="ref39">39</xref>] and SynSys [<xref ref-type="bibr" rid="ref40">40</xref>] produce scalable, transparent timelines but reflect population-level distributions rather than individual patient data. EHR-grounded approaches improve realism. Generative Adversarial Network (GAN) variants [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>] address missing features and mixed data types, achieving higher fidelity than rule-based methods on MIMIC-III/IV benchmarks [<xref ref-type="bibr" rid="ref43">43</xref>], while SimSUM [<xref ref-type="bibr" rid="ref44">44</xref>] combines Bayesian networks with prompted LLMs to synthesize clinical notes. Systems integrating EHR grounding with conversational retrieval, including ophthalmology simulators [<xref ref-type="bibr" rid="ref38">38</xref>] and multiagent knowledge graph pipelines [<xref ref-type="bibr" rid="ref45">45</xref>], advance clinical accuracy but treat the encounter as a transactional data exchange rather than a dynamic, rapport-dependent interaction.</p></sec><sec id="s1-2-5"><title>Behavioral and Linguistic Modeling</title><p>Beyond medical profile construction, systems must express profiles through natural dialogue with appropriate communication style and conversational dynamics [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref46">46</xref>]. Cognitive and persona-based approaches model psychological states to generate behaviorally coherent conversation. PATIENT &#x03A8; [<xref ref-type="bibr" rid="ref47">47</xref>] uses expert-designed cognitive schemas with GPT-4 to reproduce emotional fluctuations and resistant behaviors, while SFMSS [<xref ref-type="bibr" rid="ref48">48</xref>] embeds Big Five traits to shape dialogue tone and coordinates patient, nurse, and supervisor agents to enforce outpatient triage workflows. PAL [<xref ref-type="bibr" rid="ref49">49</xref>] simulates emotionally nuanced palliative care patient interactions with NURSE-framework feedback, illustrating affect-grounded simulation in another condition-specific clinical context. These systems remain condition-specific (eg, mental health and palliative care) and lack medical grounding for clinical evaluation. Prompt-based approaches [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref37">37</xref>] offer scalability but lack persistent patient state tracking and systematic control over linguistic and behavioral variation.</p></sec><sec id="s1-2-6"><title>Procedural and Workflow Control</title><p>A third class prioritizes procedural fidelity and clinical workflows through discrete agent roles and actions. iPDG [<xref ref-type="bibr" rid="ref50">50</xref>] enforces clinical plausibility through manually defined domain rules. Such systems achieve procedural coherence but sacrifice conversational flexibility and behavioral depth.</p></sec><sec id="s1-2-7"><title>Limitations and Gaps</title><p>Recent systematic reviews of mental health chatbots found that LLM-based systems concentrate in early stage technical validation, with most evaluations focused on conversational quality or adherence to specific prompts in controlled settings rather than rigorous testing for clinical benefit [<xref ref-type="bibr" rid="ref51">51</xref>]. Existing patient simulation systems exhibit 2 critical gaps that constrain trustworthy AI evaluation. First, no existing framework integrates medical realism, behavioral variation, and linguistic diversity within a unified architecture. Recent systems pair medical grounding with behavioral modeling or behavioral depth with linguistic variation, but this fragmentation creates evaluation blind spots. Existing systems cannot systematically test whether agents maintain diagnostic accuracy across complex comorbidities, varied health literacy, and adversarial behavior simultaneously. Multiturn failures like context loss and inconsistent safety emerge precisely at these intersections.</p><p>Second, existing approaches prioritize medical outcomes over systematic risk management aligned with frameworks like the NIST AI RMF. They lack explicit risk mapping connecting simulation parameters to risk categories, auditability with traceable lineage to source data, and controllable risk probing through systematic variation. This prevents the detection of consequential failures: agents may recommend different treatments when patients express identical conditions at different health literacy levels, or safety mechanisms may fail under adversarial conditions that only systematic variation can expose. We address both gaps through a unified framework whose medical, linguistic, and behavioral profiles align with the NIST AI RMF Map and Measure functions, enabling risk assessment across medical accuracy, communication appropriateness, and behavioral robustness.</p></sec></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>The framework integrated a patient simulator with an AI decision aid to enable systematic evaluation of conversational clinical decision-making. The patient simulator produced controlled, profile-driven responses by combining medical, linguistic, and behavioral characteristics, while the AI decision aid conducted a structured conversational intake to elicit clinical history and generate antidepressant recommendations. <xref ref-type="fig" rid="figure1">Figure 1</xref> illustrates this interaction. This section details the patient simulator design, AI decision aid, and evaluation methodology.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>A schematic diagram of the conversation between the patient simulator and the AI decision aid system. LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e100772_fig01.png"/></fig><p>The research team integrated clinical expertise in psychiatric and mental health nursing with expertise in clinical informatics, health administration, and natural language processing, with clinical authors contributing to the patient profiles, antidepressant selection task, and evaluation framework. The broader Patient-Centered Outcomes Research Institute (PCORI)&#x2013;funded project was guided by an advisory board of clinicians, mental health organization leaders, and individuals with lived experience of depression.</p></sec><sec id="s2-2"><title>Data</title><sec id="s2-2-1"><title>Overview</title><p>Both the patient simulator medical profiles and the AI decision aid prediction models were derived from the All of Us Research Program Registered Tier v8 dataset, a national longitudinal study providing structured EHR data, including conditions, medications, procedures, and demographics [<xref ref-type="bibr" rid="ref17">17</xref>].</p></sec><sec id="s2-2-2"><title>Cohort Selection</title><p>We extracted data from All of Us participants with major depressive disorder (Systematized Nomenclature of Medicine Clinical Terms [SNOMED CT] code 370143000 and its descendants). The resulting cohort included 58,446 participants, contributing 466,752 antidepressant trials, with 18,471 diagnoses, 2642 medications, and 5001 procedures. Outcomes captured responses to 14 antidepressants and 1 category covering all remaining antidepressants (dataset cutoff: October 1, 2023). Antidepressant response was defined as taking the antidepressant for at least 10 weeks without switching or augmenting with another antidepressant [<xref ref-type="bibr" rid="ref52">52</xref>].</p></sec><sec id="s2-2-3"><title>Data Application</title><p>For patient simulator medical profiles, the dataset provided feature distributions, RR calculations for outcome-relevant predictor selection, and demographic distributions for age and gender initialization. The simulator was evaluated on the 4 most common antidepressants: fluoxetine, sertraline, trazodone, and duloxetine [<xref ref-type="bibr" rid="ref53">53</xref>]. For the AI decision aid, the same data trained prediction models generating treatment recommendations across all 14 antidepressants. All data use complied with All of Us dissemination policies.</p></sec></sec><sec id="s2-3"><title>Patient Simulator Design</title><sec id="s2-3-1"><title>Design Rationale and NIST Alignment</title><p>The patient simulator design was grounded in the NIST AI RMF, which defines four core functions: (1) <italic>Govern</italic> (accountability and oversight), (2) <italic>Map</italic> (risk identification), (3) <italic>Measure</italic> (risk assessment), and (4) <italic>Manage</italic> (risk mitigation) [<xref ref-type="bibr" rid="ref16">16</xref>]. We aligned with Map and Measure across 2 complementary phases. Simulator construction supported Map by structuring risk identification through 3 profile dimensions, each addressing a distinct category of risk: medical profiles targeting clinical accuracy, linguistic profiles targeting communication-dependent risks, and behavioral profiles targeting interaction-dependent risks. Simulator-AI interaction supported Measure by surfacing how AI performance varies across that structured profile space, facilitating downstream risk characterization.</p></sec><sec id="s2-3-2"><title>Medical Profile Requirements</title><p>Medical profiles provided the clinical foundation for risk assessment, requiring realistic and diverse medical contexts to evaluate performance across clinical heterogeneity. Generating such profiles from structured EHR data faces interconnected challenges: high-dimensional feature spaces (10&#x2075;-10&#x2076; concept codes) with sparse task-relevant signals, data quality issues (missingness and inconsistent coding), and medically implausible feature combinations [<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref56">56</xref>]. Our design adhered to five AI RMF&#x2013;aligned trustworthiness requirements: (1) <italic>controllability</italic> (emphasize task-relevant features while maintaining outcome diversity), (2) <italic>coherence</italic> (maintain clinical plausibility across diagnoses, treatments, and temporal events, including rare but valid scenarios), (3) <italic>variability</italic> (capture heterogeneous comorbidities and contextual factors to expose brittleness and subgroup bias), (4) <italic>efficiency</italic> (balance clinical completeness with computational tractability for large-scale evaluation), and (5) <italic>transparency</italic> (maintain traceable feature lineage to source distributions for targeted risk analysis). The mapping of these requirements to data challenges and mitigated risks is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-3-3"><title>Linguistic and Behavioral Profile Requirements</title><p>Medical profiles establish clinical context but cannot capture communication-dependent and interaction-dependent risks, requiring 2 additional profile dimensions, consistent with the safety taxonomy proposed by Lim et al [<xref ref-type="bibr" rid="ref57">57</xref>]. These are needed because accurate medical retrieval can still fail due to how a patient communicates or how the conversation unfolds. The 2 dimensions map onto the patient input types and hazardous scenarios mentioned by Lim et al [<xref ref-type="bibr" rid="ref57">57</xref>], informing the communication-dependent and interaction-dependent categories used here.</p><p>Communication-dependent risks emerge from heterogeneity in patient expression. Agents must maintain safety and comprehensibility across health literacy levels, condition-specific patterns, and vernacular variations [<xref ref-type="bibr" rid="ref58">58</xref>]. Linguistic profiles enable controlled assessment grounded in health literacy and psycholinguistic research [<xref ref-type="bibr" rid="ref59">59</xref>].</p><p>Interaction-dependent risks emerge when patients manipulate information, test boundaries, or withhold details [<xref ref-type="bibr" rid="ref60">60</xref>]. Because evaluation cannot assess safety mechanism robustness without systematic behavioral variation, behavioral profiles are grounded in clinical and human-computer interaction research to probe agent resilience across conditions.</p></sec><sec id="s2-3-4"><title>Implementation</title><p>The patient simulator integrated 3 profiles: medical profiles grounded in EHR data, linguistic profiles capturing health literacy and condition-specific variation, and behavioral profiles representing engagement and interaction patterns. Profiles were operationalized through structured LLM prompting, where the medical profile assigned hierarchical indices to each clinical fact (eg, [3.2] Individual Psychotherapy). For each intake question, the model identified relevant indexed facts; applied linguistic style transfer (eg, [3.2] &#x201C;talked to someone&#x201D;); and constructed a JSON response containing the original facts, style-transferred equivalents, and a final natural language response with inline span markers. Simulator-provided markers were removed before being passed to the AI decision aid to ensure independent extraction. The complete prompt is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-3-5"><title>Medical Profiles</title><p>Phase 1 generated outcome-relevant, coherent, and diverse patient profiles. Phase 2 applied probabilistic selection using a binomial distribution derived from All of Us antidepressant response data, ensuring that the final cohort reflects realistic population-level variability.</p></sec></sec><sec id="s2-4"><title>Medical Profile Generation (Phase 1)</title><sec id="s2-4-1"><title>Overview</title><p>Profile generation used three core design principles [<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref62">62</xref>]: (1) prioritize outcome-relevant features, (2) enforce statistical independence among selected features, and (3) inject controlled diversity from residual features, implemented through 4 stages (relevance filtering, demographic initialization, independence screening, and diversity expansion), as outlined in <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Abbreviated medical profile generation algorithm. Complete algorithm details are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e100772_fig02.png"/></fig></sec><sec id="s2-4-2"><title>Stage 1: Top-K Filtering</title><p>The algorithm restricted feature candidates to the top K predictors of the antidepressant response outcome, <italic>e</italic><sub>0</sub> (K=500), improving sample efficiency and aligning with the max-relevance component of minimum redundancy maximum relevance feature selection [<xref ref-type="bibr" rid="ref62">62</xref>].</p></sec><sec id="s2-4-3"><title>Stage 2: Demographic Seeding</title><p>Each profile <italic>S</italic> was initialized with age and gender from All of Us demographic distributions.</p></sec><sec id="s2-4-4"><title>Stage 3: Independence-Screened Selection</title><p>Features were added iteratively, with each candidate <italic>v</italic> screened for statistical independence from already-selected features. The RR <italic>RR</italic>(<italic>s</italic>,<italic>v</italic>) quantifies the association between features <italic>s</italic> and <italic>v</italic> with respect to antidepressant response:</p><disp-formula id="E1"><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>R</mml:mi><mml:mi>R</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>v</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow></mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x2223;</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>s</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x2229;</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>v</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow></mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x2223;</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>s</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where values near 1 indicate independence, values &#x003E;1.5 indicate positive association, and values &#x003C;0.67 indicate negative association. Each candidate was required to satisfy:</p><disp-formula id="E2"><mml:math id="eqn2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>1.5</mml:mn><mml:mo>&#x003C;</mml:mo><mml:mi>R</mml:mi><mml:mi>R</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>v</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2264;</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">g</mml:mi><mml:mi mathvariant="normal">h</mml:mi></mml:mrow></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mi>s</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>S</mml:mi></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>This symmetric band excluded near-deterministic couplings (<italic>RR</italic> &#x003E;high) and strong anticorrelations (<italic>RR</italic> &#x003C;1/1.5), ensuring statistically independent and clinically coherent feature combinations [<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref62">62</xref>]. The upper threshold (high) was set to 7.</p></sec><sec id="s2-4-5"><title>Stage 4: Diversity Expansion</title><p>After forming a coherent feature set, residual features outside the top-K set were added if they satisfied:</p><disp-formula id="E3"><mml:math id="eqn3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>R</mml:mi><mml:mi>R</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>u</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x003E;</mml:mo><mml:mn>1.5</mml:mn><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2203;</mml:mi><mml:mi>s</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>S</mml:mi></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>This captured latent contextual variables enriching patient heterogeneity without compromising coherence [<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref62">62</xref>]. The complete pseudocode is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec></sec><sec id="s2-5"><title>Probabilistic Patient Selection (Phase 2)</title><sec id="s2-5-1"><title>Overview</title><p>Phase 2 selected patients whose response probabilities reproduce the All of Us population-level distribution by dividing the probability range into 7 sigma bands of a binomial distribution <italic>B</italic>(<italic>n</italic>,<italic>p</italic>), with sampling weighted to match expected binomial frequencies. For example, with <italic>n</italic>=100 and <italic>p</italic>=0.4 (<italic>&#x03BC;</italic>=40; <italic>&#x03C3;</italic>&#x2248;4.9), approximately 64% of patients fall within (<italic>&#x03BC;</italic>-<italic>&#x03C3;</italic>, <italic>&#x03BC;</italic>+<italic>&#x03C3;</italic>), 15% in each adjacent band (<italic>&#x03BC;</italic>&#x00B1;1<italic>&#x03C3;</italic> to <italic>&#x03BC;</italic>&#x00B1;2<italic>&#x03C3;</italic>), and &#x003C;3% in the tails beyond <italic>&#x03BC;</italic>&#x00B1;2<italic>&#x03C3;</italic>. This stratified sampling preserves central tendency and natural variability while avoiding over- or underrepresentation of extreme responders. Detailed sigma-band allocations appear in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><p>For AI RMF alignment, <italic>controllability</italic> and <italic>coherence</italic> were achieved through outcome-relevant feature selection and RR screening, captured by the proportion of outcome-related features and the average RR. <italic>Variability</italic> was assessed via the number of unique concept codes, <italic>efficiency</italic> was determined by the average features per profile, and <italic>transparency</italic> was evaluated through explicit feature lineage enabling auditable profile construction.</p><p>The framework had 2 tunable hyperparameters governing profile density: the high threshold from stage 3 and the number of residual features added in stage 4 (set here to 3&#x2010;5 per profile). Lowering <italic>high</italic> narrowed the set of candidate concepts eligible for inclusion. Varying it across {4,5,6} yielded 5&#x2010;8 features per profile on average versus 8&#x2010;9 at <italic>high</italic>=7, a gradual rather than sharp change indicating limited sensitivity to the exact threshold within this range. Either hyperparameter can be adjusted to target a desired profile density.</p></sec><sec id="s2-5-2"><title>Linguistic Profiles</title><p>Conversational agents must maintain safety and comprehensibility across diverse patient expression styles [<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref64">64</xref>]. Systematic linguistic variation exposes blind spots and failure modes [<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref66">66</xref>]. The simulator implemented a dual-axis linguistic framework with profiles along two independent dimensions: (1) a health literacy gradient capturing variation in comprehension, terminology, and discourse structure [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref67">67</xref>,<xref ref-type="bibr" rid="ref68">68</xref>], and (2) condition-specific communication reflecting linguistic patterns characteristic of depression and anxiety disorders derived from Linguistic Inquiry and Word Count (LIWC)&#x2013;based clinical analyses [<xref ref-type="bibr" rid="ref69">69</xref>]. Health literacy generalizes across clinical tasks as comprehension barriers affect patient-agent interaction regardless of medical condition. Condition-specific profiles capture diagnostic patterns (<italic>depression</italic> and <italic>anxiety</italic>) adaptable to other conditions. <xref ref-type="table" rid="table1">Table 1</xref> specifies the 5 linguistic profiles.</p><p>The linguistic profiles align with NIST requirements through literature-grounded linguistic variation [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref67">67</xref>-<xref ref-type="bibr" rid="ref69">69</xref>], enabling controllable assessment of communication-dependent risks.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Linguistic user profiles across multiple dimensions.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Profile</td><td align="left" valign="top">Key linguistic attributes</td><td align="left" valign="top">Example response</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Health literacy</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Limited</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Style: Concrete, informal, sometimes vague</p></list-item><list-item><p>Tone: Hesitant, uncertain, conversational</p></list-item><list-item><p>Vocab: Everyday terms, slang, vague quantities</p></list-item><list-item><p>Structure: Short, fragmented sentences; frequent fillers</p></list-item><list-item><p>Patterns: Minimal elaboration unless prompted</p></list-item></list></td><td align="left" valign="top">&#x201C;Uh, just my morning pill. you know, the one for my nerves.&#x201D;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Functional</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Style: Clear, basic descriptions of symptoms or routines</p></list-item><list-item><p>Tone: Cooperative, open</p></list-item><list-item><p>Vocab: Mix of common and medical terms</p></list-item><list-item><p>Structure: Simple narratives; occasional causal reasoning</p></list-item><list-item><p>Patterns: Provides coherent answers; asks clarifying questions</p></list-item></list></td><td align="left" valign="top">&#x201C;I take Prozac every morning. It helps my mood, but I still have trouble sleeping.&#x201D;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Proficient</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Style: Precise, clinical, well-organized</p></list-item><list-item><p>Tone<italic>:</italic> Confident, analytical</p></list-item><list-item><p>Vocab: Technical terms; qualifiers such as &#x201C;likely&#x201D; or &#x201C;seems improved&#x201D;</p></list-item><list-item><p>Structure: Multiclause, logically sequenced sentences</p></list-item><list-item><p>Patterns<italic>:</italic> References timelines; anticipates follow-up questions</p></list-item></list></td><td align="left" valign="top">&#x201C;I&#x2019;m on fluoxetine, 20 milligrams daily. It&#x2019;s effective, though I&#x2019;ve noticed mild insomnia.&#x201D;</td></tr><tr><td align="left" valign="top" colspan="3">Condition-specific</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Depression</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Style: Brief, muted, sometimes resigned</p></list-item><list-item><p>Tone: Flat, pessimistic, self-critical</p></list-item><list-item><p>Vocab: Negative emotion words; self-focused phrasing</p></list-item><list-item><p>Structure: Short, often past-tense statements</p></list-item><list-item><p>Patterns: Withdrawn responses; dismisses reassurance</p></list-item></list></td><td align="left" valign="top">&#x201C;Barely sleeping. My head won&#x2019;t shut off.&#x201D;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Illness anxiety disorder</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Style: Symptom-focused and repetitive</p></list-item><list-item><p>Tone: Anxious, urgent</p></list-item><list-item><p>Vocab: Symptom terms; &#x201C;what if&#x201D; speculation; absolutist wording</p></list-item><list-item><p>Structure: Mix of run-on sentences and abrupt alarms</p></list-item><list-item><p>Patterns<italic>:</italic> Reassurance-seeking cycles; future-oriented worry</p></list-item></list></td><td align="left" valign="top">&#x201C;I felt a flutter. What if it&#x2019;s heart failure even though the test was normal?&#x201D;</td></tr></tbody></table></table-wrap></sec><sec id="s2-5-3"><title>Behavioral Profiles</title><p>Conversational agents struggle when patients go off-topic, respond vaguely, or withhold information [<xref ref-type="bibr" rid="ref70">70</xref>,<xref ref-type="bibr" rid="ref71">71</xref>]. Systematic behavioral variation is essential for stress-testing agent robustness and assessing recovery from conversational breakdowns [<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref73">73</xref>]. We organized the 13 patient behaviors documented by Simpson et al [<xref ref-type="bibr" rid="ref74">74</xref>] into 4 behavioral categories (complete mapping is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>), with <italic>structured and cooperative</italic> as an additional fifth category. This study operationalized 3 profiles: <italic>distracted and unfocused</italic> and <italic>adversarial and combative</italic> to capture challenging dynamics, with <italic>structured and cooperative</italic> as the baseline. We selected this subset to bound experimental scope while preserving the contrast that matters for risk assessment: a cooperative baseline against 2 profiles that diverge in distinct ways. Each profile varied along 4 dimensions: conversational adherence, engagement, topical focus, and adversarial behavior, as detailed in <xref ref-type="table" rid="table2">Table 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Behavioral user profiles across multiple dimensions.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Profile</td><td align="left" valign="top">Key attributes</td><td align="left" valign="top">Example response</td></tr></thead><tbody><tr><td align="left" valign="top">Structured and cooperative</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Adherence: High</p></list-item><list-item><p>Engagement: High</p></list-item><list-item><p>Topical focus: High</p></list-item><list-item><p>Adversarial/toxic behavior: Minimal</p></list-item></list></td><td align="left" valign="top">&#x201C;Yes, I take 20 mg of fluoxetine every morning around 8 AM. I haven&#x2019;t missed a dose in the last three weeks.&#x201D;</td></tr><tr><td align="left" valign="top">Distracted and unfocused</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Adherence: Low</p></list-item><list-item><p>Engagement<italic>:</italic> Sporadic</p></list-item><list-item><p>Topical focus<italic>:</italic> Off-topic</p></list-item><list-item><p>Adversarial/toxic behavior<italic>:</italic> Inadvertent derailment of conversation</p></list-item></list></td><td align="left" valign="top">&#x201C;I was. wait, which one? Oh right, yeah I think? But yesterday I forgot &#x2014; also my dog wouldn&#x2019;t eat.&#x201D;</td></tr><tr><td align="left" valign="top">Adversarial and combative</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Adherence<italic>:</italic> Variable</p></list-item><list-item><p>Engagement<italic>:</italic> Variable</p></list-item><list-item><p>Topical focus<italic>:</italic> Variable</p></list-item><list-item><p>Adversarial/toxic behavior: Overtly confrontational or hostile</p></list-item></list></td><td align="left" valign="top">&#x201C;What kind of dumb question is that? Maybe if your system worked better, I wouldn&#x2019;t have to answer this again.&#x201D;</td></tr></tbody></table></table-wrap><p>The behavioral profiles align with NIST requirements through empirically grounded behavioral variation [<xref ref-type="bibr" rid="ref74">74</xref>], enabling controlled assessment of interaction-dependent risks including adversarial scenarios with transparent lineage.</p><p>Together, these 3 profile dimensions provide comprehensive Map and Measure coverage for conversational health care AI risk assessment.</p></sec></sec><sec id="s2-6"><title>Chain-of-Thought Prompting Strategy</title><p>We used chain-of-thought (CoT) prompting [<xref ref-type="bibr" rid="ref75">75</xref>] to generate realistic patient responses through a structured process. For each question posed by the AI decision aid, the prompt directed the model to (1) identify relevant indexed medical attributes, (2) apply controlled term-level linguistic transformations, and (3) construct a natural language response with explicit references. Behavioral constraints were enforced as rules governing interaction, while linguistic constraints shaped the form of expression. All outputs used a fixed JSON schema, enabling traceability and systematic evaluation (details are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s2-7"><title>AI Decision Aid</title><p>The AI decision aid is a multiagent conversational platform for antidepressant selection, serving as the system under evaluation in this black-box assessment. The system conducted structured intake through 6 sequential stages: establishing rapport, collecting illness history, gathering antidepressant history, documenting current medications, recording clinical procedures, and generating personalized recommendations. Each stage used LLM-guided dialogue to elicit clinical information and a Retrieval-Augmented Generation (RAG) system to normalize medical concepts before passing them to an analytical reasoning system for estimating antidepressant response. Concept normalization was a 2-step process: an embedding-based retriever returned the top-K candidate concepts from the study lexicon for each patient utterance, and an LLM selected the single best match from those candidates. Only this rank-1 selection was forwarded to the analytical model, and downstream candidates were discarded. If the correct concept was in the top K but not ranked first, it did not reach the recommendation step. The analytical advice system was based on the Direct Effects Multiplicative Inference (DEMI) algorithm [<xref ref-type="bibr" rid="ref76">76</xref>], which estimates dependent Bayesian relationships among medical concepts to support clinical reasoning under partial observability. Alternative recommendation models (eg, logistic regression) are compatible, and the system specifications and code are publicly available [<xref ref-type="bibr" rid="ref24">24</xref>].</p></sec><sec id="s2-8"><title>Evaluation Framework</title><sec id="s2-8-1"><title>Overview</title><p>The evaluation framework examined patient simulator performance across medical, linguistic, and behavioral dimensions: medical profiles through human and LLM-based annotation using a 3-label schema (accurate, inaccurate, and unsupported), and linguistic and behavioral profiles through human annotation, quantitative metrics, and visual clustering.</p></sec><sec id="s2-8-2"><title>Human Annotation</title><p>All annotations were performed by 2 annotators with complementary domain expertise: one had an undergraduate degree in psychology, and one was a registered nurse. Each sample was independently annotated by both raters, with disagreements resolved through adjudication.</p></sec><sec id="s2-8-3"><title>Medical Profile Evaluation</title><p>The patient simulator CoT process (1) identified relevant medical concepts, (2) rephrased each concept according to the linguistic profile, and (3) constructed a natural language response based on linguistic and behavioral characteristics. For each conversational turn, the simulator outputted (1) relevant concepts referenced numerically (eg, [2.3]) and (2) a natural language response with inline spans marking where each concept is expressed. These spans served as annotation units labeled using a three-category schema as follows: (1) <italic>accurate</italic>, where the medical fact is correctly expressed with minor colloquialisms (eg, &#x201C;happy pills&#x201D; for &#x201C;Prozac&#x201D;) permitted if core clinical meaning is preserved; (2) <italic>inaccurate</italic>, where the fact is present but misrepresented, distorting critical clinical details or using implausible phrasing; and (3) <italic>unsupported</italic>, where content does not correspond to the patient&#x2019;s profile, capturing hallucinated or fabricated details outside tagged spans. This approach evaluated expression fidelity conditioned on correct concept retrieval. Concept recall was computed programmatically from the simulator&#x2019;s traced outputs, with 95% of profile concepts appearing at least once in the generated conversations.</p><p>We validated the annotation and judging pipeline itself through controlled error injection. The patient simulator expressed nearly all medical concepts accurately, so annotators could achieve artificially high interannotator agreement (IAA) by labeling all concepts as <italic>accurate</italic>, and the LLM judge validation would lack discriminative power. To enable rigorous evaluation, we introduced controlled semantic perturbations by replacing clinical concepts with semantically similar but clinically distinct alternatives (eg, &#x201C;hypertension,&#x201D; &#x201C;prehypertension,&#x201D; &#x201C;diabetes mellitus,&#x201D; and &#x201C;prediabetes&#x201D;), selected via a semantic search over SNOMED CT and Current Procedural Terminology, 4th edition (CPT-4) that identified the top 20 candidates, which were randomly shuffled before ontology-based filtering removed hierarchical ancestors or descendants and enforced minimum hierarchical distance. The first qualifying candidate was selected, ensuring variation rather than defaulting to the nearest semantic neighbor; if none qualified, the threshold was relaxed or the pool expanded. The resulting perturbed profiles maintained linguistic realism while introducing subtle clinical inaccuracies for evaluating both human annotation quality and LLM judge performance.</p></sec><sec id="s2-8-4"><title>Linguistic Profile Evaluation</title><p>Linguistic profiles were evaluated through human annotation, where annotators classified each conversation into 1 of 5 predefined linguistic profiles. Five automated metrics were also used:</p><list list-type="order"><list-item><p>Reading level: Flesch-Kincaid grade level (FKGL) [<xref ref-type="bibr" rid="ref77">77</xref>] estimates the school grade level and is calculated per response turn and averaged per conversation.</p></list-item><list-item><p>Average response length: Average words per turn using NLTK&#x2019;s English tokenizer [<xref ref-type="bibr" rid="ref78">78</xref>].</p></list-item><list-item><p>Medical term density: Clinical term count via greedy n-gram matching (up to 6 grams) against the study medical lexicon (all concept codes present in the database).</p></list-item><list-item><p>Depression score: Mean turn-level probability from an XLM-RoBERTa&#x2013;based depressive symptom classifier [<xref ref-type="bibr" rid="ref79">79</xref>].</p></list-item><list-item><p>t-distributed stochastic neighbor embedding (t-SNE): Projects linguistic features to 2 dimensions using cosine distance; multiple seeds are tested to minimize Kullback-Leibler divergence.</p></list-item></list></sec><sec id="s2-8-5"><title>Behavioral Profile Evaluation</title><p>Annotators classified each conversation into 1 of 3 behavior profiles. Three automated metrics quantified behavioral patterns:</p><list list-type="order"><list-item><p>On-topic similarity: Average cosine similarity between each AI decision aid turn and the patient simulator response.</p></list-item><list-item><p>Toxicity: Mean turn-level probability from the XLM-RoBERTa&#x2013;based toxicity classifier [<xref ref-type="bibr" rid="ref80">80</xref>].</p></list-item><list-item><p>t-SNE: Projects behavioral features analogously to linguistic t-SNE.</p></list-item></list></sec><sec id="s2-8-6"><title>LLM Judge Evaluation</title><p>LLM-based judges have demonstrated strong performance in health care evaluation tasks [<xref ref-type="bibr" rid="ref81">81</xref>-<xref ref-type="bibr" rid="ref83">83</xref>]. Automated natural language processing similarity metrics, by contrast, have not been shown to correlate with expert human evaluation of generative LLM outputs on EHR data [<xref ref-type="bibr" rid="ref84">84</xref>]. We used a separate LLM, Claude Opus 4.6, as an automated judge, applying the same annotation schema across medical, linguistic, and behavioral dimensions. The judge prompt was developed on 45 held-out conversations without human annotations. Judge alignment was then validated against human annotations on the full evaluation set, with results presented in the Results section.</p></sec></sec><sec id="s2-9"><title>Experimental Paradigm</title><p>Experiments used 60 medical profiles evaluated across 4 antidepressants (sertraline: 17%, trazodone: 15%, fluoxetine: 13%, and duloxetine: 11%) reflecting common clinical practice [<xref ref-type="bibr" rid="ref53">53</xref>]. For each antidepressant, a pool of patient profiles was simulated and down-sampled to 15 profiles per drug by stratifying across response probability bands. With <italic>high</italic>=7 and 3&#x2010;5 residual features per profile, generation yielded an average of 8&#x2010;9 clinical features per profile, consistent with reports that outpatient visits address 5.4&#x2010;7.1 clinical items per encounter [<xref ref-type="bibr" rid="ref85">85</xref>]. This feature density reflects plausible patient disclosure and contrasts with All of Us records (approximately 98.88 features per patient).</p><p>The design comprised three settings: (1) 5 linguistic profiles with fixed <italic>structured and cooperative</italic> behavior (300 conversations); (2) 3 behavioral profiles with fixed <italic>functional health literacy</italic> (180 conversations); and (3) combined linguistic-behavioral variation (150 conversations), yielding 500 unique simulated conversations. The patient simulator and AI decision aid were powered by GPT-4.1. Claude Opus 4.6 served as the LLM judge, and all embedding-based analyses, including t-SNE projections and on-topic similarity, used OpenAI&#x2019;s text-embedding-3-small model, which also supports RAG-based concept normalization in the AI decision aid. Both have been implemented as separate agents within LangGraph [<xref ref-type="bibr" rid="ref86">86</xref>].</p></sec><sec id="s2-10"><title>Ethical Considerations</title><p>This project was examined by George Mason University&#x2019;s Institutional Review Board (study 2154028&#x2010;1 for work on the All of Us database and study 00000437 for evaluation of the text of the advice through annotation) and was determined to be &#x201C;not research&#x201D; in the context of the definition of research on human subjects by the United States Department of Health and Human Services. Informed consent was therefore not required. All of Us data were accessed exclusively through the program&#x2019;s secure Researcher Workbench under the data use agreement, and no individual-level records were redistributed. Annotation work was performed by named research personnel acknowledged in the manuscript. No human subjects were recruited for this study.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Simulated Medical Profile Validation</title><p>Across 60 medical profiles, the framework demonstrated systematic risk characterization aligned with the NIST AI RMF Map and Measure functions. <italic>Controllability</italic> was achieved by emphasizing outcome-related features (37.2% of features per profile) while maintaining outcome diversity across response probability bands. <italic>Coherence</italic> was reflected in an average RR of 2.64 among outcome-related features, consistent with the RR gating. <italic>Variability</italic> was maintained through 292 unique concept codes across the 60 profiles. <italic>Efficiency</italic> was demonstrated by an average of 8.08 features per profile, substantially lower than the 98.88 features in a typical All of Us record. <italic>Transparency</italic> was maintained through explicit feature provenance across all generation components: demographic attributes followed All of Us distributions (age 13&#x2010;19 years: 748/515,405, 0.1%; 20&#x2010;40 years: 130,890/515,405, 25.3%; 41&#x2010;64 years: 213,726/515,405, 41.5%; 65&#x2010;79 years: 143,775/515,405, 28.0%; 80&#x2010;89 years: 26,266/515,405, 5.1%; male: 192,379/515,405, 37.3%; female: 323,026/515,405, 62.7%), outcome relevance was enforced using the top-K set, residual features were drawn from the non&#x2013;top-K pool, and outcome probabilities matched antidepressant-specific binomial distributions (fluoxetine: range 0.28&#x2010;0.57, mean 0.41; sertraline: range 0.30&#x2010;0.59, mean 0.44; duloxetine: range 0.28&#x2010;0.57, mean 0.43; and trazodone: range 0.16&#x2010;0.42, mean 0.29).</p></sec><sec id="s3-2"><title>Medical Profile Validation</title><p>Medical profile fidelity was assessed through 1786 simulator-generated medical concepts with annotated spans across 100 conversations, including 292 perturbed concepts. <xref ref-type="table" rid="table3">Table 3</xref> shows IAA between human annotators and the LLM judge. Across all concepts, human IAA was high (&#x03BA;=0.73; <italic>F</italic><sub>1</sub>-score=0.93). Agreement decreased for perturbed concepts (&#x03BA;=0.30; <italic>F</italic><sub>1</sub>-score=0.76), reflecting the difficulty of subtle semantic substitutions. Human-LLM judge agreement followed the same pattern (overall: &#x03BA;=0.78; <italic>F</italic><sub>1</sub>-score=0.93; perturbed concepts: &#x03BA;=0.24; <italic>F</italic><sub>1</sub>-score=0.77). Annotators identified 109 unique <italic>unsupported</italic> cases, predominantly incidental medication mentions (eg, Tylenol, ibuprofen, and vitamins) and minor symptom references (eg, headache and neck pain) not present in the medical profiles (IAA: <italic>F</italic><sub>1</sub>-score=0.75). The LLM judge identified 76 <italic>unsupported</italic> cases of the same type (<italic>F</italic><sub>1</sub>-score=0.58 against adjudicated human annotations). We also performed a paired bootstrap test [<xref ref-type="bibr" rid="ref87">87</xref>] with 10,000 resamples. We found that human-human agreement (&#x03BA;=0.73) was slightly higher than agreement between individual human annotators and the LLM (&#x03BA;=0.70&#x2010;0.71), but these differences were not statistically significant (<italic>P</italic>=.21 and <italic>P</italic>=.32, respectively). Moreover, agreement between the LLM and each annotator did not differ significantly (<italic>P</italic>=.63), indicating no systematic annotator-specific bias. Overall, LLM-human agreement was comparable to human-human agreement within sampling variability.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Agreement scores for human-human and human-LLM<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> evaluations across full and perturbed concept sets.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top" colspan="2">Evaluation and concept set</td><td align="left" valign="top">Value, n</td><td align="left" valign="top">Cohen &#x03BA;</td><td align="left" valign="top">Accurate <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Inaccurate <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Unsupported <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Overall <italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="8">Human-human</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Perturbed</td><td align="left" valign="top">292</td><td align="left" valign="top">0.30</td><td align="left" valign="top">0.45</td><td align="left" valign="top">0.85</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="top">0.76</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Full</td><td align="left" valign="top">1786</td><td align="left" valign="top">0.73</td><td align="left" valign="top">0.96</td><td align="left" valign="top">0.76</td><td align="left" valign="top">0.75</td><td align="left" valign="top">0.93</td></tr><tr><td align="left" valign="top" colspan="8">Human-LLM</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Perturbed</td><td align="left" valign="top">292</td><td align="left" valign="top">0.24</td><td align="left" valign="top">0.37</td><td align="left" valign="top">0.87</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="top">0.77</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Full</td><td align="left" valign="top">1786</td><td align="left" valign="top">0.78</td><td align="left" valign="top">0.97</td><td align="left" valign="top">0.81</td><td align="left" valign="top">0.58</td><td align="left" valign="top">0.93</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table3fn2"><p><sup>b</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><p>In the LLM judge evaluation of medical profiles, of 8210 expressed concepts, 7932 (96.6%) were labeled <italic>accurate</italic>, 88 (1.1%) were labeled <italic>inaccurate</italic>, and 190 (2.3%) were labeled <italic>unsupported</italic>. The simulator maintained high accuracy (96%&#x2010;99%) and low error rates across all profiles. Within this range, concept expression frequency increased across the literacy gradient, peaking with <italic>proficient</italic>. Error types diverged modestly by profile: <italic>proficient</italic> showed the highest <italic>unsupported</italic> rate (10/981, 1.0%) due to greater medical term density, while <italic>illness anxiety disorder</italic> had the highest inaccuracy rate (30/1192, 2.5%) due to its repetitive tone. The <italic>depression</italic> profile produced the fewest expressed concepts with minimal errors.</p><p><xref ref-type="table" rid="table4">Table 4</xref> provides a breakdown of fidelity across linguistic and behavioral profiles. <italic>Structured and cooperative</italic> ensured stability, <italic>distracted and unfocused</italic> introduced frequent <italic>unsupported</italic> outputs, and <italic>adversarial and combative</italic> minimized errors, as confrontational responses tended to be terse, limiting opportunities for errors.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>LLM<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> judge evaluation of medical profile fidelity across linguistic and behavioral profiles.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top" colspan="2">Profile name and metrics</td><td align="left" valign="top">Total, n</td><td align="left" valign="top">Accurate, n (%)</td><td align="left" valign="top">Inaccurate, n (%)</td><td align="left" valign="top">Unsupported, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Linguistic profiles under the structured and cooperative behavioral condition</td><td align="left" valign="top">4830</td><td align="left" valign="top">4724 (97.8)</td><td align="left" valign="top">66 (1.4)</td><td align="left" valign="top">40 (0.8)</td></tr><tr><td align="left" valign="top" colspan="6"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Health literacy</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Limited</td><td align="left" valign="top">863</td><td align="left" valign="top">849 (98.4)</td><td align="left" valign="top">11 (1.3)</td><td align="left" valign="top">3 (0.3)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Functional</td><td align="left" valign="top">952</td><td align="left" valign="top">935 (98.2)</td><td align="left" valign="top">13 (1.4)</td><td align="left" valign="top">4 (0.4)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Proficient</td><td align="left" valign="top">981</td><td align="left" valign="top">968 (98.7)</td><td align="left" valign="top">3 (0.3)</td><td align="left" valign="top">10 (1.0)</td></tr><tr><td align="left" valign="top" colspan="6"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Condition-specific</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Depression</td><td align="left" valign="top">842</td><td align="left" valign="top">826 (98.1)</td><td align="left" valign="top">9 (1.1)</td><td align="left" valign="top">7 (0.8)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Illness anxiety disorder</td><td align="left" valign="top">1192</td><td align="left" valign="top">1146 (96.1)</td><td align="left" valign="top">30 (2.5)</td><td align="left" valign="top">16 (1.3)</td></tr><tr><td align="left" valign="top" colspan="2">Behavioral profiles under the functional health literacy linguistic condition</td><td align="left" valign="top">3035</td><td align="left" valign="top">2906 (95.7)</td><td align="left" valign="top">18 (0.6)</td><td align="left" valign="top">111 (3.7)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Structured and cooperative</td><td align="left" valign="top">952</td><td align="left" valign="top">935 (98.2)</td><td align="left" valign="top">13 (1.4)</td><td align="left" valign="top">4 (0.4)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Distracted and unfocused</td><td align="left" valign="top">988</td><td align="left" valign="top">884 (89.5)</td><td align="left" valign="top">5 (0.5)</td><td align="left" valign="top">99 (10.0)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Adversarial and combative</td><td align="left" valign="top">1095</td><td align="left" valign="top">1087 (99.3)</td><td align="left" valign="top">0 (0.0)</td><td align="left" valign="top">8 (0.7)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap><p>The intersection of linguistic and behavioral profiles shaped error type rather than overall accuracy, which remained high across all combinations (<xref ref-type="fig" rid="figure3">Figure 3</xref>). <italic>Distracted and unfocused</italic> produced elevated <italic>unsupported</italic> rates in lower health literacy profiles, where conversational drift introduced off-profile content. <italic>Adversarial and combative</italic> drove inaccuracies in <italic>proficient</italic> and <italic>illness anxiety disorder</italic> profiles, where confrontational responses distorted rather than fabricated clinical content. <italic>Structured and cooperative</italic> yielded the lowest error rates, though <italic>illness anxiety disorder</italic> produced <italic>unsupported</italic> content even under cooperative conditions.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Large language model judge evaluations from the intersection of linguistic and behavioral profiles: (A) accurate, (B) inaccurate, and (C) unsupported. HL: health literacy.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e100772_fig03.png"/></fig></sec><sec id="s3-3"><title>Linguistic Profile Validation</title><p>Human annotators showed substantial agreement on linguistic profile classification (&#x03BA;=0.61; micro <italic>F</italic><sub>1</sub>-score=0.70), comparable to human-LLM agreement (&#x03BA;=0.63; micro <italic>F</italic><sub>1</sub>-score=0.70). Adjudicated labels against predefined profiles reached a micro <italic>F</italic><sub>1</sub>-score of 0.87, indicating consistent expression of intended profiles, and the LLM judge achieved a micro <italic>F</italic><sub>1</sub>-score of 0.95 across 500 conversations.</p><p><xref ref-type="table" rid="table5">Table 5</xref> shows linguistic variation under fixed <italic>structured and cooperative</italic> behavior. Reading level increased monotonically from <italic>limited</italic> (FKGL=3.59) to <italic>proficient</italic> (FKGL=11.90), with response length and medical term density following the same gradient (<italic>limited</italic>: 34.34 words, 4.33 medical terms; <italic>proficient</italic>: 38.91 words, 11.68 medical terms). The <italic>depression</italic> profile produced the shortest responses (21.48 words; depression score=0.20), whereas <italic>illness anxiety disorder</italic> produced the longest responses (65.05 words; medical term density=14.27; depression score=0.25), reflecting greater health-related elaboration.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Evaluation of linguistic profiles under the structured and cooperative behavioral condition.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top" colspan="2">Profile name and metrics</td><td align="left" valign="top">Reading level (FKGL<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>)</td><td align="left" valign="top">Response length</td><td align="left" valign="top">Medical term density</td><td align="left" valign="top">Depression score</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">Health literacy</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Limited</td><td align="left" valign="top">3.59</td><td align="left" valign="top">34.34</td><td align="left" valign="top">4.33</td><td align="left" valign="top">0.01</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Functional</td><td align="left" valign="top">7.63</td><td align="left" valign="top">28.70</td><td align="left" valign="top">9.83</td><td align="left" valign="top">0.00</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Proficient</td><td align="left" valign="top">11.90</td><td align="left" valign="top">38.91</td><td align="left" valign="top">11.68</td><td align="left" valign="top">0.00</td></tr><tr><td align="left" valign="top" colspan="6">Condition-specific</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Depression</td><td align="left" valign="top">4.49</td><td align="left" valign="top">21.48</td><td align="left" valign="top">6.17</td><td align="left" valign="top">0.20</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Illness anxiety disorder</td><td align="left" valign="top">8.11</td><td align="left" valign="top">65.05</td><td align="left" valign="top">14.27</td><td align="left" valign="top">0.25</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>FKGL: Flesch-Kincaid grade level.</p></fn></table-wrap-foot></table-wrap><p><xref ref-type="fig" rid="figure4">Figure 4</xref> shows t-SNE clustering where linguistic profiles form distinct clusters, with <italic>functional</italic> and <italic>proficient</italic> overlapping due to their shared characteristics, and <italic>depression</italic> partially overlapping with lower literacy profiles. Conversely, <italic>illness anxiety disorder</italic> formed a distinct cluster. Together, these findings confirm graded, distinct linguistic profiles.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>t-distributed stochastic neighbor embedding visualization of response embeddings for linguistic profiles under the structured and cooperative behavioral condition (A) and behavioral profiles under the functional health literacy linguistic condition (B).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e100772_fig04.png"/></fig></sec><sec id="s3-4"><title>Behavioral Profile Validation</title><p>Human annotators showed high agreement on behavioral profile classification (&#x03BA;=0.93; micro <italic>F</italic><sub>1</sub>-score=0.96), identical to human-LLM agreement, with classification against predefined profiles reaching a micro <italic>F</italic><sub>1</sub>-score of 0.98. Under fixed <italic>functional health literacy</italic> (<xref ref-type="table" rid="table6">Table 6</xref>), <italic>structured and cooperative</italic> achieved baseline on-topic similarity (0.51) and the lowest toxicity (0.0003). <italic>Distracted and unfocused</italic> showed slightly reduced on-topic similarity (0.50), consistent with conversational drift, while maintaining low toxicity (0.0035). <italic>Adversarial and combative</italic> exhibited substantially higher toxicity (0.0564) while retaining the highest topical relevance (0.52).</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Evaluation of behavioral profiles under the functional health literacy linguistic condition.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Profile</td><td align="left" valign="top">On-topic similarity</td><td align="left" valign="top">Toxicity</td></tr></thead><tbody><tr><td align="left" valign="top">Structured and cooperative</td><td align="left" valign="top">0.51</td><td align="left" valign="top">0.0003</td></tr><tr><td align="left" valign="top">Distracted and unfocused</td><td align="left" valign="top">0.50</td><td align="left" valign="top">0.0035</td></tr><tr><td align="left" valign="top">Adversarial and combative</td><td align="left" valign="top">0.52</td><td align="left" valign="top">0.0564</td></tr></tbody></table></table-wrap><p><xref ref-type="fig" rid="figure4">Figure 4</xref> shows t-SNE clustering with clear separation among behavioral profiles. Together, these findings confirm distinct behavioral profiles under fixed linguistic conditions.</p></sec><sec id="s3-5"><title>Profile Interactions</title><p><xref ref-type="fig" rid="figure5">Figure 5</xref> presents intersectional evaluation across linguistic and behavioral profile combinations. <italic>Distracted and unfocused</italic> produced the longest responses across all linguistic profiles (61&#x2010;82 words vs 21&#x2010;66 for <italic>structured and cooperative</italic>), while <italic>depression</italic> consistently produced the shortest responses (21&#x2010;32 words) with elevated depression scores (0.18&#x2010;0.25). Toxicity showed the strongest behavioral dominance, with <italic>adversarial and combative</italic> exhibiting elevated toxicity (range: 0.03&#x2010;0.11) regardless of linguistic profile, whereas reading level and depression scores remained largely determined by linguistic profiles. On-topic similarity showed modest behavioral effects, with <italic>structured and cooperative</italic> achieving slightly higher alignment than <italic>distracted and unfocused</italic>. These patterns demonstrate that the simulator maintains profile independence where intended (linguistic features preserved across behaviors) while capturing realistic interactions (behavioral effects on engagement and tone).</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Intersection of linguistic and behavioral profiles across evaluation metrics: (A) reading level, (B) response length, (C) medical jargon, (D) depression score, (E) on-topic similarity, and (F) toxicity. HL: health literacy.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e100772_fig05.png"/></fig></sec><sec id="s3-6"><title>AI Decision Aid Performance: Risk Measurement Across Patient Variation</title><sec id="s3-6-1"><title>Overview</title><p>This section evaluates AI decision aid performance across 500 simulated conversations.</p></sec><sec id="s3-6-2"><title>Concept Retrieval Coverage</title><p>Overall concept recall reached 93% (n=2622) across 2819 reference concepts, indicating that approximately 7% (n=197) were not identified by the AI decision aid. Recall was the highest for diagnoses (1958/2077, 94.3%) and procedures (518/569, 91.0%), while medications showed the lowest recall (146/173, 84.4%), identifying medication retrieval as a relative vulnerability. The system additionally introduced 151 concepts outside predefined profiles, predominantly medications (n=84) and procedures (n=57), reflecting conversational elaboration.</p></sec><sec id="s3-6-3"><title>Retrieval Accuracy and Health Literacy Effects</title><p>The retrieval performance of concepts identified by the AI decision aid was reported on the full set of simulator-generated reference concepts (n=2819), with unidentified concepts contributing to overall outcomes. Overall rank-1 accuracy was 65.9% (1857/2819); however, 80.2% (2261/2819) of concepts appeared within the top 20 candidates, indicating ranking imprecision rather than missing semantic coverage. Retrieval accuracy increased with health literacy level. Rank-1 accuracy ranged from 47.6% (<italic>limited</italic>) to 81.9% (<italic>proficient</italic>). Condition-specific profiles fell in between, with <italic>depression</italic> at 62.0% and <italic>illness anxiety disorder</italic> at 63.7%. Vague or affect-heavy language reduced ranking accuracy (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p><p>Patients with <italic>limited</italic> health literacy or <italic>depression</italic> used colloquial, fragmented, or affect-heavy language that misaligned with the system&#x2019;s canonical concept vocabulary. Because top-20 retrieval did not fully close this gap, the system introduces a structural bias: retrieval is reliable for patients who use precise medical terminology but degrades for populations expressing equivalent clinical content through less structured language (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p></sec><sec id="s3-6-4"><title>Downstream Risk to Antidepressant Recommendation</title><p>Recommendations were produced by DEMI over 15 antidepressants plus a <italic>no recommendation</italic> category (probability &#x003C;0.1), making downstream performance sensitive to information loss during intake. Performance improved with higher health literacy, with <italic>proficient</italic> achieving the highest <italic>F</italic><sub>1</sub>-score (0.73) under s<italic>tructured and cooperative</italic> and <italic>limited</italic> achieving the lowest <italic>F</italic><sub>1</sub>-score (0.48). Behavioral profiles further modulated this risk: <italic>structured and cooperative</italic> yielded more stable outcomes, while <italic>distracted and unfocused</italic> and <italic>adversarial and combative</italic> were associated with performance shifts that varied by linguistic profile, as shown in <xref ref-type="fig" rid="figure6">Figure 6</xref>.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p><italic>F</italic><sub>1</sub>-scores for antidepressant recommendation before and after AI decision aid processing across linguistic and behavioral profile combinations. HL: health literacy.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e100772_fig06.png"/></fig></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Overview</title><p>This study demonstrated that patient simulation, grounded in the NIST AI RMF, can systematically expose risks in conversational health care AI that static evaluation methods miss. Across 500 simulated conversations, the framework revealed a monotonic performance gradient across health literacy levels, with corresponding effects on antidepressant recommendation accuracy. The remainder of this section discusses these principal findings, their clinical implications, and the limitations of the work.</p></sec><sec id="s4-2"><title>Principal Findings</title><sec id="s4-2-1"><title>Medical Profile Generation</title><p>The medical profile generation process produced clinically coherent profiles with explicit feature lineage traceable to source EHR data. To validate both annotator and LLM judge performance, we applied controlled error injection. Semantically similar but clinically distinct perturbations (eg, hypertension &#x2192; prehypertension, diabetes mellitus &#x2192; prediabetes, major depressive disorder moderate &#x2192; major depressive disorder mild, and generalized anxiety disorder &#x2192; adjustment disorder with anxiety) tested annotator and LLM judge sensitivity. The injection was challenging by design. Each perturbation is semantically close to the original concept, the kind of subtle substitution an LLM is likely to produce, and once rendered in a patient&#x2019;s communication style, it is easily mistaken for an accurate expression. Agreement on perturbed concepts was lower than overall agreement (human-human: &#x03BA;=0.30 vs &#x03BA;=0.73; human-LLM: &#x03BA;=0.24 vs &#x03BA;=0.78). This partly reflects the intended difficulty of the perturbations, which approximate real-world concept ambiguity, but also signals a real detection limit relevant to deployment monitoring. If human raters and the LLM judge both struggle to catch subtle clinical substitutions under these controlled conditions, comparable errors are unlikely to be easier to catch in live use, where similar ambiguity could go undetected without dedicated monitoring.</p></sec><sec id="s4-2-2"><title>Linguistic and Behavioral Profile Effectiveness</title><p>The simulator produced measurably distinct profiles across both dimensions. Behavioral profiles formed categorically different clusters, while linguistic profiles formed a continuous health literacy gradient. The <italic>functional</italic> and <italic>proficient</italic> overlap reflects their genuine similarity along this continuum. Cross-dimensional analysis confirmed profile independence with 2 interaction effects: <italic>distracted and unfocused</italic> consistently increased response length, and <italic>adversarial and combative</italic> increased toxicity regardless of linguistic profile.</p></sec><sec id="s4-2-3"><title>Health Literacy as a Risk Factor for Clinical AI</title><p>The linguistic profile substantially affected AI decision aid performance, with monotonic degradation across the health literacy gradient: rank-1 retrieval ranged from 47.6% (<italic>limited</italic>) to 69.6% (<italic>functional</italic>) and to 81.9% (<italic>proficient</italic>), with the gradient persisting in downstream antidepressant recommendation accuracy. The system relied on precise terminology: colloquial expressions like &#x201C;my morning pill for my nerves&#x201D; force the retrieval system to resolve a larger semantic distance than &#x201C;20 mg of fluoxetine for generalized anxiety disorder.&#x201D; These failures point to specific Manage-function interventions: clarification prompting, terminology normalization, or multipass retrieval. Condition-specific profiles showed intermediate retrieval performance (<italic>depression</italic>: 62.0%, <italic>illness anxiety disorder</italic>: 63.7%), suggesting that affective expression also degrades performance, though less severely than low health literacy.</p></sec><sec id="s4-2-4"><title>LLM Judge Viability</title><p>The LLM judge aligned strongly with human annotators on medical concept agreement (overall <italic>F</italic><sub>1</sub>-score=0.93) and showed lower agreement on perturbed concepts (<italic>F</italic><sub>1</sub>-score=0.77), mirroring divergence among human annotators on those same items. The judge also flagged <italic>unsupported</italic> content beyond the perturbations, related to incidental medications and minor symptoms not present in the profiles. These results support LLM judge use for large-scale screening, with human review reserved for edge cases.</p></sec></sec><sec id="s4-3"><title>Comparison With Prior Work</title><p>Patient simulators have advanced rapidly, yet none of these simulators have jointly addressed medical, linguistic, and behavioral risks within a single risk-aligned framework. PATIENT-&#x03A8; [<xref ref-type="bibr" rid="ref47">47</xref>] achieves strong behavioral and emotional realism through 106 expert-authored cognitive behavioral therapy cognitive schemas, with a focus on clinical training rather than EHR-grounded risk assessment. PatientSim [<xref ref-type="bibr" rid="ref35">35</xref>] benchmarks LLM doctor agents using MIMIC-derived clinical profiles and a 4-axis persona space (personality, language proficiency, recall, and confusion), with persona treated as a fixed categorical attribute and traceability provided at the persona level rather than the concept level. MATRIX [<xref ref-type="bibr" rid="ref57">57</xref>] audits hazards across 2100 dialogues using a safety-engineered hazard taxonomy, a simulated patient (PatBot), and an LLM judge, with clinical content drawn from scenario templates and a self-contained safety taxonomy rather than alignment to an external risk-management framework. Cook et al [<xref ref-type="bibr" rid="ref34">34</xref>] demonstrated training scalability through prompt-engineered GPT virtual patients on 2 outpatient topics, with the evaluation directed at clinician history-taking rather than at the AI system itself. In contrast, our patient simulator unifies 3 risk-aligned profile dimensions in a single architecture: EHR-grounded medical profiles drawn from All of Us via RR gating, 5 linguistic profiles spanning 3 health literacy levels (<italic>limited</italic>, <italic>functional</italic>, and <italic>proficient</italic>) and 2 condition-specific styles (<italic>depression</italic> and <italic>illness anxiety disorder</italic>), and 3 behavioral profiles (<italic>structured and cooperative</italic>, <italic>distracted and unfocused</italic>, and <italic>adversarial and combative</italic>). The framework aligns with the NIST AI RMF Map and Measure functions and provides concept-level lineage from each utterance back to its source clinical facts.</p></sec><sec id="s4-4"><title>Clinical Implications</title><sec id="s4-4-1"><title>Overview</title><p>Performance differences across health literacy and behavioral profiles raise clinical, equity, and governance concerns that should be addressed before the AI decision aid is deployed.</p></sec><sec id="s4-4-2"><title>Health Literacy as a Patient Safety Issue</title><p>Rank-1 retrieval fell from 81.9% (<italic>proficient</italic>) to 47.6% (<italic>limited</italic>). Because antidepressant trials typically last 6-12 weeks before effectiveness can be assessed [<xref ref-type="bibr" rid="ref88">88</xref>,<xref ref-type="bibr" rid="ref89">89</xref>], retrieval errors may not surface until after weeks of ineffective or harmful treatment. <italic>Limited health literacy</italic> is more common among patients with severe depression, lower socioeconomic status, advanced age, and less formal education [<xref ref-type="bibr" rid="ref90">90</xref>,<xref ref-type="bibr" rid="ref91">91</xref>], which are the same groups that bear the heaviest burden of untreated mental illness [<xref ref-type="bibr" rid="ref92">92</xref>,<xref ref-type="bibr" rid="ref93">93</xref>]. The performance gap is concentrated in the population most in need of accessible support. This is not a technical limitation to be addressed in a later release; it is an immediate patient safety risk requiring active mitigation before deployment.</p></sec><sec id="s4-4-3"><title>Behavioral Profiles Reflect Psychiatric Phenotypes</title><p>The <italic>distracted and unfocused</italic> profile mirrors the psychomotor slowing, impaired attention, and disorganized thought of moderate-to-severe depression [<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref95">95</xref>], which are core symptoms of the illness rather than mere variations in communication style. The <italic>adversarial and combative</italic> profile often reflects trauma history, prior negative health care experiences, or treatment frustration rather than a difficult personality [<xref ref-type="bibr" rid="ref96">96</xref>,<xref ref-type="bibr" rid="ref97">97</xref>]. The system is least reliable for patients whose illness severity itself makes communication difficult. These clinical interpretations explain why the observed performance gaps carry safety significance; the behavioral profiles were derived from a general interaction taxonomy [<xref ref-type="bibr" rid="ref74">74</xref>] and have not been validated against symptom presentations aligned with the Diagnostic and Statistical Manual of Mental Disorders (DSM). Thus, the mapping to specific psychiatric phenotypes should be read as illustrative rather than diagnostic.</p></sec><sec id="s4-4-4"><title>Care Pathway Integration Determines Harm Potential</title><p>Provider-supervised use makes the identified gaps manageable, and direct-to-patient deployment, particularly to underserved populations with limited provider access, amplifies them. Deployment plans should specify the oversight model, review criteria, and escalation pathways for low-confidence recommendations. Under the NIST AI RMF, these governance structures correspond directly to Manage-function commitments and should be specified before deployment approval. Where the system informs treatment selection, providers should also interpret its outputs as patterns of medication persistence, recognizing that persistence and symptom remission can diverge in real-world treatment, with prognostic implications [<xref ref-type="bibr" rid="ref98">98</xref>]. The Manage-function interventions (clarification prompting, terminology normalization, and adaptive retrieval) should be developed with input from patients with limited health literacy and providers experienced in psychiatric communication, and validated in the same simulation framework before rollout.</p></sec><sec id="s4-4-5"><title>Equity and Participatory Design</title><p>The performance gradient documented here is a familiar pattern in health care AI. Tools designed and validated on data representing dominant communication styles reinforce existing disparities at deployment scale rather than reducing them [<xref ref-type="bibr" rid="ref99">99</xref>]. Antidepressant selection compounds this risk. Comorbidity, pharmacogenomics, and social determinants all shape outcomes, and patients with limited health literacy already face diagnostic and treatment gaps in standard psychiatric care [<xref ref-type="bibr" rid="ref100">100</xref>,<xref ref-type="bibr" rid="ref101">101</xref>]. A system that underperforms for these populations risks widening those gaps, and technical correction alone will not address this. The identified Manage-function interventions should be co-designed with patients carrying limited health literacy and with clinicians who serve them. Psychiatric care has developed practices for this work, including teach-back protocols, plain language standards, and culturally adapted patient education [<xref ref-type="bibr" rid="ref102">102</xref>]. Conversational AI for these populations should be built on the same foundation.</p></sec></sec><sec id="s4-5"><title>Limitations</title><sec id="s4-5-1"><title>Generalizability</title><p>Simulated conversations have not been validated against real patient interactions. Ecological validity will be evaluated in a planned real-world trial. The evaluation covers a single clinical task and 4 antidepressants. The framework itself does not assume a deployment model, but the harm potential of the performance gaps we report depends on how the AI decision aid is deployed. Future evaluations should match the oversight model under which it will be used.</p></sec><sec id="s4-5-2"><title>Validity of Clinical Signals</title><p>Antidepressant response was operationalized as 10-week medication persistence without switching or augmentation, a surrogate influenced by formulary access, provider inertia, and barriers to follow-up, which does not capture clinical remission [<xref ref-type="bibr" rid="ref103">103</xref>,<xref ref-type="bibr" rid="ref104">104</xref>]. Future work should anchor response in patient-reported outcomes or validated symptom scales (eg, Patient Health Questionnaire-9 and Hamilton Rating Scale for Depression) [<xref ref-type="bibr" rid="ref105">105</xref>,<xref ref-type="bibr" rid="ref106">106</xref>].</p></sec><sec id="s4-5-3"><title>Profile Construction</title><p>Medical profiles relied exclusively on structured EHR data, omitting the unstructured clinical narrative where clinicians document symptom severity, treatment history, and psychosocial context. The independence screening criterion may exclude valid coupled comorbidities, underrepresenting patients with strongly correlated conditions. Behavioral profiles follow the taxonomy by Simpson et al [<xref ref-type="bibr" rid="ref74">74</xref>] and are not validated against DSM-aligned presentations: the <italic>distracted and unfocused</italic> profile resembles depressive neurovegetative symptoms but is not clinically verified, and the <italic>adversarial and combative</italic> profile does not distinguish oppositional engagement from trauma- or frustration-driven presentations. Validation against psychiatrically characterized populations remains a priority.</p></sec><sec id="s4-5-4"><title>Model Dependence</title><p>The patient simulator and AI decision aid both used GPT-4.1, though information leakage between them was structurally prevented. The simulator&#x2019;s internal profile and concept markers were stripped before its response reached the AI decision aid, which extracted clinical information from natural language text alone, as it would from a real patient. A shared model family could still produce correlated blind spots not present with 2 independently trained systems. The LLM judge (Claude Opus 4.6) was a different model, mitigating this risk at the evaluation layer but not for the simulator-decision aid pairing. Future work should test decision aid performance against simulator output using an alternative model family on a subsample.</p></sec><sec id="s4-5-5"><title>Annotation Diversity</title><p>All annotations were performed by 2 raters. While their complementary clinical backgrounds (psychology and nursing) and adjudicated agreement provide some assurance, a 2-rater design limits the generalizability of the reported agreement estimates and cannot capture the full range of expert interpretation. Future evaluations should incorporate a larger, more diverse annotator pool.</p></sec></sec><sec id="s4-6"><title>Conclusions</title><p>Applied to a conversational antidepressant decision aid, our patient simulation framework found a steep gap in performance across health literacy levels. Rank-1 concept retrieval was 81.9% for proficient health literacy and 47.6% for limited literacy. Recommendation accuracy followed the same pattern. This gap is important, as limited health literacy is common among patients with the most severe psychiatric illness. The patients least served by this system are those who need it the most. Identifying this risk required simulated patients grounded across all 3 dimensions: medical, linguistic, and behavioral. Prior simulators have varied 1 or 2 of these dimensions, but not all 3 together. While we evaluated the framework only on antidepressant selection, extending it to other clinical decision-aid tasks remains a direction for future work. The primary contributions include a RR-based algorithm for auditable medical profile generation, a controlled perturbation method for validating annotators, and an LLM judge with human-level agreement. Scalable simulation and evaluation are also enabled, and thus, candidate mitigations can be developed and tested against the same risks before deployment. However, technical fixes for AI systems are not sufficient. Health systems must specify care pathways, oversight protocols, and escalation requirements. Moreover, they must clinically validate the system in the populations it currently underserves.</p></sec></sec></body><back><ack><p>The study used data from the All of Us Research Program&#x2019;s Registered Tier Dataset v8, available to all authorized users on the Researcher Workbench. We gratefully acknowledge All of Us participants for their contributions, without whom this research would not have been possible. We also thank the National Institutes of Health&#x2019;s All of Us Research Program for making available the participant data examined in this study. We also thank the members of the project advisory board, comprising clinicians, leaders of national mental health organizations, and individuals with lived experience of depression, for their ongoing guidance throughout the Patient-Centered Outcomes Research Institute&#x2013;funded research program.</p><p>We thank Francisca Mainoo, MPH, BSN, RN, and Roya Minovi, BA, for their careful annotation work supporting the evaluation reported in this study.</p><p>The authors used Claude Opus 4.7 (Anthropic) and GPT-5.5 (OpenAI), accessed through their respective web interfaces, for language polishing during manuscript preparation. The tools were not used to generate scientific content, perform analysis, or produce references. The authors reviewed all suggestions and retain full responsibility for the manuscript.</p></ack><notes><sec><title>Funding</title><p>This work was supported by a Patient-Centered Outcomes Research Institute (PCORI) Award (ME-2024C1-36732). The views in this article are solely the responsibility of the authors and do not necessarily represent the views of the PCORI, its board of governors, or the methodology committee.</p></sec><sec><title>Data Availability</title><p>Medical profiles were generated using risk ratio gating based on electronic health record data from the All of Us Research Program Registered Tier Dataset v8. Antidepressant response probabilities were computed using the Direct Effects Multiplicative Inference (DEMI) algorithm based on the same dataset. The risk ratio database and DEMI model weights will be released with publication (participant-level All of Us records cannot be redistributed under the program&#x2019;s data use agreement). All code, simulated patient profiles, conversation logs, and evaluation annotations are available on GitHub [<xref ref-type="bibr" rid="ref24">24</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: MTRS, KPE, FA, KL</p><p>Data curation: MTRS, MSI, HRAE, KRR, YL, VFC, FA, KL</p><p>Formal analysis: MTRS, MSI, KRR, YL, FA, KL</p><p>Funding acquisition: KPE, FA, KL</p><p>Investigation: MTRS, KPE, FA, KL</p><p>Methodology: MTRS, FA, KL</p><p>Project administration: VFC, FA, KL</p><p>Resources: MTRS, MSI, VFC</p><p>Supervision: FA, KL</p><p>Validation: MTRS, MSI, KRR, YL, VFC, FA, KL</p><p>Visualization: MTRS</p><p>Writing &#x2013; original draft: MTRS, KL</p><p>Writing &#x2013; review &#x0026; editing: HRAE, VFC, KPE, FA</p><p>All authors reviewed and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>FA has a pending patent (#19/253,342) assigned to George Mason University related to the Direct Effects Multiplicative Inference methods described in this work. The other authors declare no conflicts of interest.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AI RMF</term><def><p>AI Risk Management Framework</p></def></def-item><def-item><term id="abb2">CoT</term><def><p>chain-of-thought</p></def></def-item><def-item><term id="abb3">CPT-4</term><def><p>Current Procedural Terminology, 4th edition</p></def></def-item><def-item><term id="abb4">DEMI</term><def><p>Direct Effects Multiplicative Inference</p></def></def-item><def-item><term id="abb5">DSM</term><def><p>Diagnostic and Statistical Manual of Mental Disorders</p></def></def-item><def-item><term id="abb6">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb7">FKGL</term><def><p> Flesch-Kincaid grade level</p></def></def-item><def-item><term id="abb8">GAN</term><def><p>Generative Adversarial Network</p></def></def-item><def-item><term id="abb9">IAA</term><def><p>interannotator agreement</p></def></def-item><def-item><term id="abb10">LIWC</term><def><p>Linguistic Inquiry and Word Count</p></def></def-item><def-item><term id="abb11">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb12">NIST</term><def><p>National Institute of Standards and Technology</p></def></def-item><def-item><term id="abb13">PCORI</term><def><p> Patient-Centered Outcomes Research Institute</p></def></def-item><def-item><term id="abb14">RAG</term><def><p>Retrieval-Augmented Generation</p></def></def-item><def-item><term id="abb15">RQ</term><def><p>research question</p></def></def-item><def-item><term id="abb16">RR</term><def><p>risk ratio</p></def></def-item><def-item><term id="abb17">SNOMED CT</term><def><p>Systematized Nomenclature of Medicine Clinical Terms</p></def></def-item><def-item><term id="abb18">t-SNE</term><def><p> t-distributed stochastic neighbor embedding</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Laranjo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Dunn</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Tong</surname><given-names>HL</given-names> </name><etal/></person-group><article-title>Conversational agents in healthcare: a systematic review</article-title><source>J Am Med Inform Assoc</source><year>2018</year><month>09</month><day>1</day><volume>25</volume><issue>9</issue><fpage>1248</fpage><lpage>1258</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocy072</pub-id><pub-id pub-id-type="medline">30010941</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maity</surname><given-names>S</given-names> </name><name name-style="western"><surname>Saikia</surname><given-names>MJ</given-names> </name></person-group><article-title>Large language models in healthcare and medical applications: a review</article-title><source>Bioengineering (Basel)</source><year>2025</year><month>06</month><day>10</day><volume>12</volume><issue>6</issue><fpage>631</fpage><pub-id pub-id-type="doi">10.3390/bioengineering12060631</pub-id><pub-id pub-id-type="medline">40564447</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSJ</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Schaekermann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Palepu</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Towards conversational diagnostic artificial intelligence</article-title><source>Nature</source><year>2025</year><month>06</month><volume>642</volume><issue>8067</issue><fpage>442</fpage><lpage>450</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-08866-7</pub-id><pub-id pub-id-type="medline">40205050</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kwame</surname><given-names>A</given-names> </name><name name-style="western"><surname>Petrucka</surname><given-names>PM</given-names> </name></person-group><article-title>A literature-based study of patient-centered care and communication in nurse-patient interactions: barriers, facilitators, and the way forward</article-title><source>BMC Nurs</source><year>2021</year><month>09</month><day>3</day><volume>20</volume><issue>1</issue><fpage>158</fpage><pub-id pub-id-type="doi">10.1186/s12912-021-00684-2</pub-id><pub-id pub-id-type="medline">34479560</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wynia</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Osborn</surname><given-names>CY</given-names> </name></person-group><article-title>Health literacy and communication quality in health care organizations</article-title><source>J Health Commun</source><year>2010</year><volume>15 Suppl 2</volume><issue>Suppl 2</issue><fpage>102</fpage><lpage>115</lpage><pub-id pub-id-type="doi">10.1080/10810730.2010.499981</pub-id><pub-id pub-id-type="medline">20845197</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Gourabathina</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gerych</surname><given-names>W</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ghassemi</surname><given-names>M</given-names> </name></person-group><article-title>The medium is the message: how non-clinical information shapes clinical decisions in LLMs</article-title><conf-name>2025 ACM Conference on Fairness, Accountability, and Transparency</conf-name><conf-date>Jun 23-26, 2025</conf-date><pub-id pub-id-type="doi">10.1145/3715275.3732121</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hua</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>W</given-names> </name><name name-style="western"><surname>Bates</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Standardizing and scaffolding health care AI-chatbot evaluation: systematic review</article-title><source>JMIR AI</source><year>2025</year><month>11</month><day>7</day><volume>4</volume><issue>1</issue><fpage>e69006</fpage><pub-id pub-id-type="doi">10.2196/69006</pub-id><pub-id pub-id-type="medline">41202290</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Templin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Fort</surname><given-names>S</given-names> </name><name name-style="western"><surname>Padmanabham</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Framework for bias evaluation in large language models in healthcare settings</article-title><source>NPJ Digit Med</source><year>2025</year><month>07</month><day>7</day><volume>8</volume><issue>1</issue><fpage>414</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01786-w</pub-id><pub-id pub-id-type="medline">40624264</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nadarzynski</surname><given-names>T</given-names> </name><name name-style="western"><surname>Knights</surname><given-names>N</given-names> </name><name name-style="western"><surname>Husbands</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Achieving health equity through conversational AI: a roadmap for design and implementation of inclusive chatbots in healthcare</article-title><source>PLOS Digit Health</source><year>2024</year><month>05</month><volume>3</volume><issue>5</issue><fpage>e0000492</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000492</pub-id><pub-id pub-id-type="medline">38696359</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wornow</surname><given-names>M</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Thapa</surname><given-names>R</given-names> </name><etal/></person-group><article-title>The shaky foundations of large language models and foundation models for electronic health records</article-title><source>NPJ Digit Med</source><year>2023</year><month>07</month><day>29</day><volume>6</volume><issue>1</issue><fpage>135</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00879-8</pub-id><pub-id pub-id-type="medline">37516790</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Guan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bian</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>B</given-names> </name><name name-style="western"><surname>Lou</surname><given-names>JG</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>H</given-names> </name></person-group><article-title>Evaluating LLM-based agents for multi-turn conversations: a survey</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 28, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2503.22458</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kwan</surname><given-names>WC</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>MT-eval: a multi-turn capabilities evaluation benchmark for large language models</article-title><conf-name>2024 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 12-16, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.1124</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Balog</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zhai</surname><given-names>C</given-names> </name></person-group><article-title>User simulation for evaluating information access systems</article-title><conf-name>2023 Annual International ACM SIGIR Conference on Research and Development in Information Retrieval in the Asia Pacific Region</conf-name><conf-date>Nov 26-28, 2023</conf-date><pub-id pub-id-type="doi">10.1145/3624918.3629549</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kalinich</surname><given-names>M</given-names> </name><name name-style="western"><surname>Luccarelli</surname><given-names>J</given-names> </name><name name-style="western"><surname>Moss</surname><given-names>F</given-names> </name><name name-style="western"><surname>Torous</surname><given-names>J</given-names> </name></person-group><article-title>Leveraging simulation to provide a practical framework for assessing the novel scope of risk of LLMs in healthcare</article-title><source>medRxiv</source><comment>Preprint posted online on  Nov 13, 2025</comment><pub-id pub-id-type="doi">10.1101/2025.11.10.25339903</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Tabassi</surname><given-names>E</given-names> </name></person-group><article-title>Artificial intelligence risk management framework (AI RMF 1.0)</article-title><source>NIST</source><year>2023</year><access-date>2025-10-28</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.nist.gov/publications/artificial-intelligence-risk-management-framework-ai-rmf-10">https://www.nist.gov/publications/artificial-intelligence-risk-management-framework-ai-rmf-10</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>All of Us Research Program Investigators</collab><name name-style="western"><surname>Denny</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Rutter</surname><given-names>JL</given-names> </name><etal/></person-group><article-title>The &#x201C;all of us&#x201D; research program</article-title><source>N Engl J Med</source><year>2019</year><month>08</month><day>15</day><volume>381</volume><issue>7</issue><fpage>668</fpage><lpage>676</lpage><pub-id pub-id-type="doi">10.1056/NEJMsr1809937</pub-id><pub-id pub-id-type="medline">31412182</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="book"><person-group person-group-type="author"><collab>Institute of Medicine (US) Committee on Health Literacy</collab></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Nielsen-Bohlman</surname><given-names>L</given-names> </name><name name-style="western"><surname>Panzer</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Kindig</surname><given-names>DA</given-names> </name></person-group><source>Health Literacy: A Prescription to End Confusion</source><year>2004</year><access-date>2026-08-02</access-date><publisher-name>National Academies Press (US)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/books/NBK216032/">https://www.ncbi.nlm.nih.gov/books/NBK216032/</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nutbeam</surname><given-names>D</given-names> </name></person-group><article-title>Health literacy as a public health goal: a challenge for contemporary health education and communication strategies into the 21st century</article-title><source>Health Promot Int</source><year>2000</year><month>09</month><day>1</day><volume>15</volume><issue>3</issue><fpage>259</fpage><lpage>267</lpage><pub-id pub-id-type="doi">10.1093/heapro/15.3.259</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pennebaker</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Mehl</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Niederhoffer</surname><given-names>KG</given-names> </name></person-group><article-title>Psychological aspects of natural language. use: our words, our selves</article-title><source>Annu Rev Psychol</source><year>2003</year><volume>54</volume><fpage>547</fpage><lpage>577</lpage><pub-id pub-id-type="doi">10.1146/annurev.psych.54.101601.145041</pub-id><pub-id pub-id-type="medline">12185209</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rude</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gortner</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Pennebaker</surname><given-names>J</given-names> </name></person-group><article-title>Language use of depressed and depression-vulnerable college students</article-title><source>Cogn Emot</source><year>2004</year><month>12</month><volume>18</volume><issue>8</issue><fpage>1121</fpage><lpage>1133</lpage><pub-id pub-id-type="doi">10.1080/02699930441000030</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Roter</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hall</surname><given-names>JA</given-names> </name></person-group><source>Doctors Talking with Patients/Patients Talking with Doctors</source><year>2006</year><publisher-name>Bloomsbury Publishing</publisher-name><pub-id pub-id-type="other">9780275990145</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Street</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Makoul</surname><given-names>G</given-names> </name><name name-style="western"><surname>Arora</surname><given-names>NK</given-names> </name><name name-style="western"><surname>Epstein</surname><given-names>RM</given-names> </name></person-group><article-title>How does communication heal? Pathways linking clinician-patient communication to health outcomes</article-title><source>Patient Educ Couns</source><year>2009</year><month>03</month><volume>74</volume><issue>3</issue><fpage>295</fpage><lpage>301</lpage><pub-id pub-id-type="doi">10.1016/j.pec.2008.11.015</pub-id><pub-id pub-id-type="medline">19150199</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><article-title>Lybargerlanguagelab/Patient-Simulation</article-title><source>GitHub</source><access-date>2026-08-19</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/lybargerlanguagelab/Patient-Simulation">https://github.com/lybargerlanguagelab/Patient-Simulation</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lan</surname><given-names>X</given-names> </name><name name-style="western"><surname>Li</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Large language models empowered agent-based modeling and simulation: a survey and perspectives</article-title><source>Humanit Soc Sci Commun</source><year>2024</year><volume>11</volume><issue>1</issue><fpage>1259</fpage><pub-id pub-id-type="doi">10.1057/s41599-024-03611-3</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Montaner</surname><given-names>M</given-names> </name><name name-style="western"><surname>L&#x00F3;pez</surname><given-names>B</given-names> </name><name name-style="western"><surname>de la Rosa</surname><given-names>JL</given-names> </name></person-group><article-title>Evaluation of recommender systems through simulated users</article-title><conf-name>6th International Conference on Enterprise Information Systems</conf-name><conf-date>Apr 14-17, 2004</conf-date><pub-id pub-id-type="doi">10.5220/0002622703030308</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Halpern</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Thain</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Measuring recommender system effects with simulated users</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 12, 2021</comment><pub-id pub-id-type="doi">10.48550/arXiv.2101.04526</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>X</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name></person-group><article-title>DuetSim: building user simulator with dual large language models for task-oriented dialogues</article-title><conf-name>2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation</conf-name><conf-date>May 20-25, 2024</conf-date><pub-id pub-id-type="doi">10.63317/5cp2zxmyham6</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schatzmann</surname><given-names>J</given-names> </name><name name-style="western"><surname>Weilhammer</surname><given-names>K</given-names> </name><name name-style="western"><surname>Stuttle</surname><given-names>M</given-names> </name><name name-style="western"><surname>Young</surname><given-names>S</given-names> </name></person-group><article-title>A survey of statistical user simulation techniques for reinforcement-learning of dialogue management strategies</article-title><source>Knowl Eng Rev</source><year>2006</year><month>06</month><volume>21</volume><issue>2</issue><fpage>97</fpage><lpage>126</lpage><pub-id pub-id-type="doi">10.1017/S0269888906000944</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Labhishetty</surname><given-names>S</given-names> </name></person-group><article-title>Models and evaluation of user simulation in information retrieval</article-title><source>Illinois University Library</source><year>2023</year><access-date>2025-11-29</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ideals.illinois.edu/items/127324">https://www.ideals.illinois.edu/items/127324</ext-link></comment></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhai</surname><given-names>C</given-names> </name></person-group><article-title>Information retrieval evaluation as search simulation: a general formal framework for IR evaluation</article-title><conf-name>2017 ACM SIGIR International Conference on Theory of Information Retrieval</conf-name><conf-date>Oct 1-4, 2017</conf-date><pub-id pub-id-type="doi">10.1145/3121050.3121070</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yun</surname><given-names>T</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>E</given-names> </name><name name-style="western"><surname>Safdari</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Sleepless nights, sugary days: creating synthetic users with health conditions for realistic coaching agent interactions</article-title><conf-name>Findings of the Association for Computational Linguistics: ACL 2025</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.findings-acl.729</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cho</surname><given-names>YM</given-names> </name><name name-style="western"><surname>Rai</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ungar</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sedoc</surname><given-names>J</given-names> </name><name name-style="western"><surname>Guntuku</surname><given-names>SC</given-names> </name></person-group><article-title>An integrative survey on mental health conversational agents to bridge computer science and medical perspectives</article-title><source>Proc Conf Empir Methods Nat Lang Process</source><year>2023</year><month>12</month><volume>2023</volume><fpage>11346</fpage><lpage>11369</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.698</pub-id><pub-id pub-id-type="medline">38618627</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cook</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Overgaard</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pankratz</surname><given-names>VS</given-names> </name><name name-style="western"><surname>Del Fiol</surname><given-names>G</given-names> </name><name name-style="western"><surname>Aakre</surname><given-names>CA</given-names> </name></person-group><article-title>Virtual patients using large language models: scalable, contextualized simulation of clinician-patient dialogue with feedback</article-title><source>J Med Internet Res</source><year>2025</year><month>04</month><day>4</day><volume>27</volume><fpage>e68486</fpage><pub-id pub-id-type="doi">10.2196/68486</pub-id><pub-id pub-id-type="medline">39854611</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kyung</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>H</given-names> </name><name name-style="western"><surname>Bae</surname><given-names>S</given-names> </name><etal/></person-group><article-title>PatientSim: a persona-driven simulator for realistic doctor-patient interactions</article-title><source>arXiv</source><comment>Preprint posted online on  May 23, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.17818</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Lei</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A survey on medical large language models: technology, application, trustworthiness, and future directions</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 6, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2406.03712</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Holderried</surname><given-names>F</given-names> </name><name name-style="western"><surname>Stegemann-Philipps</surname><given-names>C</given-names> </name><name name-style="western"><surname>Herrmann-Werner</surname><given-names>A</given-names> </name><etal/></person-group><article-title>A language model-powered simulated patient with automated feedback for history taking: prospective study</article-title><source>JMIR Med Educ</source><year>2024</year><month>08</month><day>16</day><volume>10</volume><fpage>e59213</fpage><pub-id pub-id-type="doi">10.2196/59213</pub-id><pub-id pub-id-type="medline">39150749</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Bi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Pang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A large language model digital patient system enhances ophthalmology history taking skills</article-title><source>NPJ Digit Med</source><year>2025</year><month>08</month><day>4</day><volume>8</volume><issue>1</issue><fpage>502</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01841-6</pub-id><pub-id pub-id-type="medline">40760042</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Walonoski</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kramer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Nichols</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Synthea: an approach, method, and software mechanism for generating synthetic patients and the synthetic electronic health care record</article-title><source>J Am Med Inform Assoc</source><year>2018</year><month>03</month><day>1</day><volume>25</volume><issue>3</issue><fpage>230</fpage><lpage>238</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocx079</pub-id><pub-id pub-id-type="medline">29025144</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dahmen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cook</surname><given-names>D</given-names> </name></person-group><article-title>SynSys: a synthetic data generation system for healthcare applications</article-title><source>Sensors (Basel)</source><year>2019</year><month>03</month><day>8</day><volume>19</volume><issue>5</issue><fpage>1181</fpage><pub-id pub-id-type="doi">10.3390/s19051181</pub-id><pub-id pub-id-type="medline">30857130</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Nyemba</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name></person-group><article-title>Generating synthetic electronic health record data using generative adversarial networks: tutorial</article-title><source>JMIR AI</source><year>2024</year><month>04</month><day>22</day><volume>3</volume><fpage>e52615</fpage><pub-id pub-id-type="doi">10.2196/52615</pub-id><pub-id pub-id-type="medline">38875595</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yoon</surname><given-names>J</given-names> </name><name name-style="western"><surname>Mizrahi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ghalaty</surname><given-names>NF</given-names> </name><etal/></person-group><article-title>EHR-Safe: generating high-fidelity and privacy-preserving synthetic electronic health records</article-title><source>NPJ Digit Med</source><year>2023</year><month>08</month><day>11</day><volume>6</volume><issue>1</issue><fpage>141</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00888-7</pub-id><pub-id pub-id-type="medline">37567968</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>X</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>H</given-names> </name><name name-style="western"><surname>Mukherjee</surname><given-names>B</given-names> </name></person-group><article-title>Generating synthetic electronic health record data: a methodological scoping review with benchmarking on phenotype data and open-source software</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>07</month><day>1</day><volume>32</volume><issue>7</issue><fpage>1227</fpage><lpage>1240</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaf082</pub-id><pub-id pub-id-type="medline">40460023</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Rabaey</surname><given-names>P</given-names> </name><name name-style="western"><surname>Heytens</surname><given-names>S</given-names> </name><name name-style="western"><surname>Demeester</surname><given-names>T</given-names> </name></person-group><article-title>SimSUM: simulated benchmark with structured and unstructured medical records</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 13, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2409.08936</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Simulated patient systems are intelligent when powered by large language model-based AI agents</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 27, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2409.18924</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Bodonhelyi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Stegemann-Philipps</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sonanini</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Modeling challenging patient interactions: LLMs for medical communication training</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 28, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2503.22250</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Milani</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chiu</surname><given-names>JC</given-names> </name><etal/></person-group><article-title>PATIENT-&#x1D713;: using large language models to simulate patients for training mental health professionals</article-title><conf-name>2024 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 12-16, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.711</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>Z</given-names> </name></person-group><article-title>SFMSS: service flow aware medical scenario simulation for conversational data generation</article-title><conf-name>Findings of the Association for Computational Linguistics: NAACL 2025</conf-name><conf-date>Apr 29 to May 4, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.findings-naacl.259</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sehgal</surname><given-names>NKR</given-names> </name><name name-style="western"><surname>Kambhamettu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ungar</surname><given-names>L</given-names> </name><name name-style="western"><surname>Guntuku</surname><given-names>SC</given-names> </name></person-group><article-title>PAL: designing conversational agents as scalable, cooperative patient simulators for palliative&#x2011;care training</article-title><conf-name>2025 Conference on Computer-Supported Cooperative Work and Social Computing</conf-name><conf-date>Oct 18-22, 2025</conf-date><pub-id pub-id-type="doi">10.1145/3715070.3749250</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Wojtusiak</surname><given-names>J</given-names> </name></person-group><article-title>Towards intelligent patient data generator</article-title><year>2016</year><access-date>2026-03-18</access-date><publisher-name>Machine Learning and Inference Laboratory, George Mason University</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.mli.gmu.edu/papers/2016/16-9.pdf">https://www.mli.gmu.edu/papers/2016/16-9.pdf</ext-link></comment></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hua</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Siddals</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Charting the evolution of artificial intelligence mental health chatbots from rule-based systems to large language models: a systematic review</article-title><source>World Psychiatry</source><year>2025</year><month>10</month><volume>24</volume><issue>3</issue><fpage>383</fpage><lpage>394</lpage><pub-id pub-id-type="doi">10.1002/wps.21352</pub-id><pub-id pub-id-type="medline">40948070</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alemi</surname><given-names>F</given-names> </name><name name-style="western"><surname>Aljuaid</surname><given-names>M</given-names> </name><name name-style="western"><surname>Durbha</surname><given-names>N</given-names> </name><etal/></person-group><article-title>A surrogate measure for patient reported symptom remission in administrative data</article-title><source>BMC Psychiatry</source><year>2021</year><month>03</month><day>4</day><volume>21</volume><issue>1</issue><fpage>121</fpage><pub-id pub-id-type="doi">10.1186/s12888-021-03133-1</pub-id><pub-id pub-id-type="medline">33663440</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="web"><article-title>Most common antidepressants</article-title><source>Definitive Healthcare</source><access-date>2025-11-11</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.definitivehc.com/resources/healthcare-insights/top-antidepressants-by-prescription-volume">https://www.definitivehc.com/resources/healthcare-insights/top-antidepressants-by-prescription-volume</ext-link></comment></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Weiskopf</surname><given-names>N</given-names> </name><name name-style="western"><surname>Abrams</surname><given-names>ZB</given-names> </name><etal/></person-group><article-title>Electronic health record data quality assessment and tools: a systematic review</article-title><source>J Am Med Inform Assoc</source><year>2023</year><month>09</month><day>25</day><volume>30</volume><issue>10</issue><fpage>1730</fpage><lpage>1740</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad120</pub-id><pub-id pub-id-type="medline">37390812</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Si</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Du</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Deep representation learning of patient data from electronic health records (EHR): a systematic review</article-title><source>J Biomed Inform</source><year>2021</year><month>03</month><volume>115</volume><fpage>103671</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2020.103671</pub-id><pub-id pub-id-type="medline">33387683</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Syed</surname><given-names>R</given-names> </name><name name-style="western"><surname>Eden</surname><given-names>R</given-names> </name><name name-style="western"><surname>Makasi</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Digital health data quality issues: systematic review</article-title><source>J Med Internet Res</source><year>2023</year><month>03</month><day>31</day><volume>25</volume><fpage>e42615</fpage><pub-id pub-id-type="doi">10.2196/42615</pub-id><pub-id pub-id-type="medline">37000497</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lim</surname><given-names>E</given-names> </name><name name-style="western"><surname>He</surname><given-names>YV</given-names> </name><name name-style="western"><surname>Joselowitz</surname><given-names>J</given-names> </name><etal/></person-group><article-title>MATRIX: multi-agent simulation framework for safe interactions and contextual clinical conversational evaluation</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 26, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2508.19163</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mo&#x00EB;ll</surname><given-names>B</given-names> </name><name name-style="western"><surname>Sand Aronsson</surname><given-names>F</given-names> </name></person-group><article-title>Harm reduction strategies for thoughtful use of large language models in the medical domain: perspectives for patients and clinicians</article-title><source>J Med Internet Res</source><year>2025</year><month>07</month><day>25</day><volume>27</volume><issue>1</issue><fpage>e75849</fpage><pub-id pub-id-type="doi">10.2196/75849</pub-id><pub-id pub-id-type="medline">40712151</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>S&#x00F8;rensen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Van den Broucke</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fullam</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Health literacy and public health: a systematic review and integration of definitions and models</article-title><source>BMC Public Health</source><year>2012</year><month>01</month><day>25</day><volume>12</volume><fpage>80</fpage><pub-id pub-id-type="doi">10.1186/1471-2458-12-80</pub-id><pub-id pub-id-type="medline">22276600</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>X</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Brahman</surname><given-names>F</given-names> </name><etal/></person-group><article-title>HAICOSYSTEM: an ecosystem for sandboxing safety risks in human-AI interactions</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 24, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2409.16427</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guyon</surname><given-names>I</given-names> </name><name name-style="western"><surname>Elisseeff</surname><given-names>A</given-names> </name></person-group><article-title>An introduction to variable and feature selection</article-title><source>J Mach Learn Res</source><year>2003</year><volume>3</volume><fpage>1157</fpage><lpage>1182</lpage><pub-id pub-id-type="doi">10.5555/944919.944968</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peng</surname><given-names>H</given-names> </name><name name-style="western"><surname>Long</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>C</given-names> </name></person-group><article-title>Feature selection based on mutual information: criteria of max-dependency, max-relevance, and min-redundancy</article-title><source>IEEE Trans Pattern Anal Mach Intell</source><year>2005</year><month>08</month><volume>27</volume><issue>8</issue><fpage>1226</fpage><lpage>1238</lpage><pub-id pub-id-type="doi">10.1109/TPAMI.2005.159</pub-id><pub-id pub-id-type="medline">16119262</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cevasco</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Morrison Brown</surname><given-names>RE</given-names> </name><name name-style="western"><surname>Woldeselassie</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kaplan</surname><given-names>S</given-names> </name></person-group><article-title>Patient engagement with conversational agents in health applications 2016-2022: a systematic review and meta-analysis</article-title><source>J Med Syst</source><year>2024</year><month>04</month><day>10</day><volume>48</volume><issue>1</issue><fpage>40</fpage><pub-id pub-id-type="doi">10.1007/s10916-024-02059-x</pub-id><pub-id pub-id-type="medline">38594411</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Wolf</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Bailey</surname><given-names>SC</given-names> </name></person-group><article-title>The role of health literacy in patient safety</article-title><source>PSNet</source><year>2009</year><access-date>2026-02-11</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://psnet.ahrq.gov/perspective/role-health-literacy-patient-safety">https://psnet.ahrq.gov/perspective/role-health-literacy-patient-safety</ext-link></comment></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cherif</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Learning diverse attacks on large language models for robust red-teaming and safety tuning</article-title><source>arXiv</source><comment>Preprint posted online on  May 28, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2405.18540</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Perez</surname><given-names>E</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Song</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Red teaming language models with language models</article-title><conf-name>2022 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Dec 7-11, 2022</conf-date><pub-id pub-id-type="doi">10.18653/v1/2022.emnlp-main.225</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="web"><article-title>National action plan to improve health literacy</article-title><source>CDC</source><year>2024</year><access-date>2025-11-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/health-literacy/php/develop-plan/national-action-plan.html">https://www.cdc.gov/health-literacy/php/develop-plan/national-action-plan.html</ext-link></comment></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paasche-Orlow</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Wolf</surname><given-names>MS</given-names> </name></person-group><article-title>The causal pathways linking health literacy to health outcomes</article-title><source>Am J Health Behav</source><year>2007</year><volume>31 Suppl 1</volume><fpage>S19</fpage><lpage>26</lpage><pub-id pub-id-type="doi">10.5555/ajhb.2007.31.supp.S19</pub-id><pub-id pub-id-type="medline">17931132</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tausczik</surname><given-names>YR</given-names> </name><name name-style="western"><surname>Pennebaker</surname><given-names>JW</given-names> </name></person-group><article-title>The psychological meaning of words: LIWC and computerized text analysis methods</article-title><source>J Lang Soc Psychol</source><year>2010</year><month>03</month><volume>29</volume><issue>1</issue><fpage>24</fpage><lpage>54</lpage><pub-id pub-id-type="doi">10.1177/0261927X09351676</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>DeVault</surname><given-names>D</given-names> </name><name name-style="western"><surname>Artstein</surname><given-names>R</given-names> </name><name name-style="western"><surname>Benn</surname><given-names>G</given-names> </name><etal/></person-group><article-title>SimSensei kiosk: a virtual human interviewer for healthcare decision support</article-title><conf-name>2014 International Conference on Autonomous Agents and Multi-Agent Systems</conf-name><conf-date>May 5-9, 2014</conf-date><pub-id pub-id-type="doi">10.5555/2615731.2617415</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fitzpatrick</surname><given-names>KK</given-names> </name><name name-style="western"><surname>Darcy</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vierhile</surname><given-names>M</given-names> </name></person-group><article-title>Delivering cognitive behavior therapy to young adults with symptoms of depression and anxiety using a fully automated conversational agent (Woebot): a randomized controlled trial</article-title><source>JMIR Ment Health</source><year>2017</year><month>06</month><day>6</day><volume>4</volume><issue>2</issue><fpage>e19</fpage><pub-id pub-id-type="doi">10.2196/mental.7785</pub-id><pub-id pub-id-type="medline">28588005</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Meng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Automatic interactive evaluation for large language models with state aware patient simulator</article-title><source>SSRN</source><comment>Preprint posted online on  Jul 15, 2024</comment><pub-id pub-id-type="doi">10.2139/ssrn.4890649</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sanjeewa</surname><given-names>R</given-names> </name><name name-style="western"><surname>Iyer</surname><given-names>R</given-names> </name><name name-style="western"><surname>Apputhurai</surname><given-names>P</given-names> </name><name name-style="western"><surname>Wickramasinghe</surname><given-names>N</given-names> </name><name name-style="western"><surname>Meyer</surname><given-names>D</given-names> </name></person-group><article-title>Empathic conversational agent platform designs and their evaluation in the context of mental health: systematic review</article-title><source>JMIR Ment Health</source><year>2024</year><month>09</month><day>9</day><volume>11</volume><issue>1</issue><fpage>e58974</fpage><pub-id pub-id-type="doi">10.2196/58974</pub-id><pub-id pub-id-type="medline">39250799</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Simpson</surname><given-names>S</given-names> </name><name name-style="western"><surname>McDowell</surname><given-names>A</given-names> </name></person-group><source>The Clinical Interview: Skills for More Effective Patient Encounters</source><year>2019</year><publisher-name>Routledge</publisher-name><pub-id pub-id-type="doi">10.4324/9780429437243</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Schuurmans</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Chain-of-thought prompting elicits reasoning in large language models</article-title><conf-name>36th International Conference on Neural Information Processing System</conf-name><conf-date>Nov 28 to Dec 9, 2022</conf-date><pub-id pub-id-type="doi">10.5555/3600270.3602070</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Alemi</surname><given-names>F</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Elyazori</surname><given-names>HRA</given-names> </name><name name-style="western"><surname>Cardenas</surname><given-names>VF</given-names> </name><name name-style="western"><surname>Ramezani</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lybarger</surname><given-names>K</given-names> </name></person-group><article-title>Causal AI reasoning: direct effects multiplicative inference algorithm</article-title><source>SSRN</source><comment>Preprint posted online on  Apr 21, 2026</comment><pub-id pub-id-type="doi">10.2139/ssrn.6537875</pub-id></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Kincaid</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Fishburne</surname><given-names>RP</given-names> </name><name name-style="western"><surname>Rogers</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Chissom</surname><given-names>BS</given-names> </name></person-group><article-title>Derivation of new readability formulas (automated readability index, fog count and Flesch reading ease formula) for navy enlisted personnel</article-title><year>1975</year><access-date>2026-02-10</access-date><publisher-name>Institute for Simulation and Training</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://stars.library.ucf.edu/cgi/viewcontent.cgi?article=1055&#x0026;context=istlibrary">https://stars.library.ucf.edu/cgi/viewcontent.cgi?article=1055&#x0026;context=istlibrary</ext-link></comment></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Loper</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bird</surname><given-names>S</given-names> </name></person-group><article-title>NLTK: the natural language toolkit</article-title><conf-name>ACL-02 Workshop on Effective Tools and Methodologies for Teaching Natural Language Processing and Computational Linguistics</conf-name><conf-date>Jul 7, 2022</conf-date><pub-id pub-id-type="doi">10.3115/1118108.1118117</pub-id></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Salazar</surname><given-names>A</given-names> </name></person-group><article-title>XLM-RoBERTa depression detection model</article-title><source>GitHub</source><access-date>2026-09-18</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/malexandersalazar/xlm-roberta-base-cls-depression">https://github.com/malexandersalazar/xlm-roberta-base-cls-depression</ext-link></comment></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bevendorff</surname><given-names>J</given-names> </name><name name-style="western"><surname>Casals</surname><given-names>XB</given-names> </name><name name-style="western"><surname>Chulvi</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Overview of PAN 2024: multi-author writing style analysis, multilingual text detoxification, oppositional thinking analysis, and generative AI authorship verification &#x2014; extended abstract</article-title><conf-name>Advances in Information Retrieval: 46th European Conference on Information Retrieval</conf-name><conf-date>Mar 24-28, 2024</conf-date><pub-id pub-id-type="doi">10.1007/978-3-031-56072-9_1</pub-id></nlm-citation></ref><ref id="ref81"><label>81</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bhat</surname><given-names>S</given-names> </name><name name-style="western"><surname>Varma</surname><given-names>V</given-names> </name></person-group><article-title>Large language models as annotators: a preliminary evaluation for annotating low-resource language content</article-title><conf-name>4th Workshop on Evaluation and Comparison of NLP Systems</conf-name><conf-date>Nov 1, 2023</conf-date><pub-id pub-id-type="doi">10.18653/v1/2023.eval4nlp-1.8</pub-id></nlm-citation></ref><ref id="ref82"><label>82</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pavlovic</surname><given-names>M</given-names> </name><name name-style="western"><surname>Poesio</surname><given-names>M</given-names> </name></person-group><article-title>The effectiveness of LLMs as annotators: a comparative overview and empirical analysis of direct representation</article-title><conf-name>3rd Workshop on Perspectivist Approaches to NLP</conf-name><conf-date>May 21, 2024</conf-date><pub-id pub-id-type="doi">10.63317/5bf5rv86yt8a</pub-id></nlm-citation></ref><ref id="ref83"><label>83</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>L</given-names> </name></person-group><article-title>LLMaAA: making large language models as active annotators</article-title><source>arXiv</source><comment>Preprint posted online on 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2310.19596</pub-id></nlm-citation></ref><ref id="ref84"><label>84</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Du</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Testing and evaluation of generative large language models in electronic health record applications: a systematic review</article-title><source>J Am Med Inform Assoc</source><year>2026</year><month>03</month><day>1</day><volume>33</volume><issue>3</issue><fpage>743</fpage><lpage>753</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaf233</pub-id><pub-id pub-id-type="medline">41528313</pub-id></nlm-citation></ref><ref id="ref85"><label>85</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abbo</surname><given-names>ED</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zelder</surname><given-names>M</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>ES</given-names> </name></person-group><article-title>The increasing number of clinical items addressed during the time of adult primary care visits</article-title><source>J Gen Intern Med</source><year>2008</year><month>12</month><volume>23</volume><issue>12</issue><fpage>2058</fpage><lpage>2065</lpage><pub-id pub-id-type="doi">10.1007/s11606-008-0805-8</pub-id><pub-id pub-id-type="medline">18830762</pub-id></nlm-citation></ref><ref id="ref86"><label>86</label><nlm-citation citation-type="web"><article-title>LangChain</article-title><source>GitHub</source><access-date>2026-05-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/langchain-ai/langchain">https://github.com/langchain-ai/langchain</ext-link></comment></nlm-citation></ref><ref id="ref87"><label>87</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Jurafsky</surname><given-names>D</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>JH</given-names> </name></person-group><source>Speech and Language Processing: An Introduction to Natural Language Processing, Computational Linguistics, and Speech Recognition with Language Models</source><year>2026</year><access-date>2026-09-18</access-date><edition>3</edition><publisher-name>Stanford University</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://web.stanford.edu/~jurafsky/slp3/">https://web.stanford.edu/~jurafsky/slp3/</ext-link></comment></nlm-citation></ref><ref id="ref88"><label>88</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Trivedi</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Rush</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Wisniewski</surname><given-names>SR</given-names> </name><etal/></person-group><article-title>Evaluation of outcomes with citalopram for depression using measurement-based care in STAR*D: implications for clinical practice</article-title><source>Am J Psychiatry</source><year>2006</year><month>01</month><volume>163</volume><issue>1</issue><fpage>28</fpage><lpage>40</lpage><pub-id pub-id-type="doi">10.1176/appi.ajp.163.1.28</pub-id><pub-id pub-id-type="medline">16390886</pub-id></nlm-citation></ref><ref id="ref89"><label>89</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rush</surname><given-names>AJ</given-names> </name></person-group><article-title>Challenges of research on treatment-resistant depression: a clinician&#x2019;s perspective</article-title><source>World Psychiatry</source><year>2023</year><month>10</month><volume>22</volume><issue>3</issue><fpage>415</fpage><lpage>417</lpage><pub-id pub-id-type="doi">10.1002/wps.21136</pub-id><pub-id pub-id-type="medline">37713554</pub-id></nlm-citation></ref><ref id="ref90"><label>90</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Degan</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Kelly</surname><given-names>PJ</given-names> </name><name name-style="western"><surname>Robinson</surname><given-names>LD</given-names> </name><name name-style="western"><surname>Deane</surname><given-names>FP</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>AM</given-names> </name></person-group><article-title>Health literacy of people living with mental illness or substance use disorders: a systematic review</article-title><source>Early Interv Psychiatry</source><year>2021</year><month>12</month><volume>15</volume><issue>6</issue><fpage>1454</fpage><lpage>1469</lpage><pub-id pub-id-type="doi">10.1111/eip.13090</pub-id><pub-id pub-id-type="medline">33254279</pub-id></nlm-citation></ref><ref id="ref91"><label>91</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paasche-Orlow</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Parker</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Gazmararian</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Nielsen-Bohlman</surname><given-names>LT</given-names> </name><name name-style="western"><surname>Rudd</surname><given-names>RR</given-names> </name></person-group><article-title>The prevalence of limited health literacy</article-title><source>J Gen Intern Med</source><year>2005</year><month>02</month><volume>20</volume><issue>2</issue><fpage>175</fpage><lpage>184</lpage><pub-id pub-id-type="doi">10.1111/j.1525-1497.2005.40245.x</pub-id><pub-id pub-id-type="medline">15836552</pub-id></nlm-citation></ref><ref id="ref92"><label>92</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lorant</surname><given-names>V</given-names> </name><name name-style="western"><surname>Deli&#x00E8;ge</surname><given-names>D</given-names> </name><name name-style="western"><surname>Eaton</surname><given-names>W</given-names> </name><name name-style="western"><surname>Robert</surname><given-names>A</given-names> </name><name name-style="western"><surname>Philippot</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ansseau</surname><given-names>M</given-names> </name></person-group><article-title>Socioeconomic inequalities in depression: a meta-analysis</article-title><source>Am J Epidemiol</source><year>2003</year><month>01</month><day>15</day><volume>157</volume><issue>2</issue><fpage>98</fpage><lpage>112</lpage><pub-id pub-id-type="doi">10.1093/aje/kwf182</pub-id><pub-id pub-id-type="medline">12522017</pub-id></nlm-citation></ref><ref id="ref93"><label>93</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Linder</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gerdtham</surname><given-names>UG</given-names> </name><name name-style="western"><surname>Trygg</surname><given-names>N</given-names> </name><name name-style="western"><surname>Fritzell</surname><given-names>S</given-names> </name><name name-style="western"><surname>Saha</surname><given-names>S</given-names> </name></person-group><article-title>Inequalities in the economic consequences of depression and anxiety in Europe: a systematic scoping review</article-title><source>Eur J Public Health</source><year>2020</year><month>08</month><day>1</day><volume>30</volume><issue>4</issue><fpage>767</fpage><lpage>777</lpage><pub-id pub-id-type="doi">10.1093/eurpub/ckz127</pub-id><pub-id pub-id-type="medline">31302703</pub-id></nlm-citation></ref><ref id="ref94"><label>94</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bennabi</surname><given-names>D</given-names> </name><name name-style="western"><surname>Vandel</surname><given-names>P</given-names> </name><name name-style="western"><surname>Papaxanthis</surname><given-names>C</given-names> </name><name name-style="western"><surname>Pozzo</surname><given-names>T</given-names> </name><name name-style="western"><surname>Haffen</surname><given-names>E</given-names> </name></person-group><article-title>Psychomotor retardation in depression: a systematic review of diagnostic, pathophysiologic, and therapeutic implications</article-title><source>Biomed Res Int</source><year>2013</year><volume>2013</volume><fpage>158746</fpage><pub-id pub-id-type="doi">10.1155/2013/158746</pub-id><pub-id pub-id-type="medline">24286073</pub-id></nlm-citation></ref><ref id="ref95"><label>95</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Park</surname><given-names>C</given-names> </name><name name-style="western"><surname>Brietzke</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Cognitive impairment in major depressive disorder</article-title><source>CNS Spectr</source><year>2019</year><month>02</month><volume>24</volume><issue>1</issue><fpage>22</fpage><lpage>29</lpage><pub-id pub-id-type="doi">10.1017/S1092852918001207</pub-id><pub-id pub-id-type="medline">30468135</pub-id></nlm-citation></ref><ref id="ref96"><label>96</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raja</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hasnain</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hoersch</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gove-Yin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Rajagopalan</surname><given-names>C</given-names> </name></person-group><article-title>Trauma informed care in medicine: current knowledge and future research directions</article-title><source>Fam Community Health</source><year>2015</year><volume>38</volume><issue>3</issue><fpage>216</fpage><lpage>226</lpage><pub-id pub-id-type="doi">10.1097/FCH.0000000000000071</pub-id><pub-id pub-id-type="medline">26017000</pub-id></nlm-citation></ref><ref id="ref97"><label>97</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mihelicova</surname><given-names>M</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shuman</surname><given-names>V</given-names> </name></person-group><article-title>Trauma-informed care for individuals with serious mental illness: an avenue for community psychology&#x2019;s involvement in community mental health</article-title><source>Am J Community Psychol</source><year>2018</year><month>03</month><volume>61</volume><issue>1-2</issue><fpage>141</fpage><lpage>152</lpage><pub-id pub-id-type="doi">10.1002/ajcp.12217</pub-id><pub-id pub-id-type="medline">29266247</pub-id></nlm-citation></ref><ref id="ref98"><label>98</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rush</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Trivedi</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Wisniewski</surname><given-names>SR</given-names> </name><etal/></person-group><article-title>Acute and longer-term outcomes in depressed outpatients requiring one or several treatment steps: a STAR*D report</article-title><source>AJP</source><year>2006</year><month>11</month><volume>163</volume><issue>11</issue><fpage>1905</fpage><lpage>1917</lpage><pub-id pub-id-type="doi">10.1176/ajp.2006.163.11.1905</pub-id></nlm-citation></ref><ref id="ref99"><label>99</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Obermeyer</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Powers</surname><given-names>B</given-names> </name><name name-style="western"><surname>Vogeli</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mullainathan</surname><given-names>S</given-names> </name></person-group><article-title>Dissecting racial bias in an algorithm used to manage the health of populations</article-title><source>Science</source><year>2019</year><month>10</month><day>25</day><volume>366</volume><issue>6464</issue><fpage>447</fpage><lpage>453</lpage><pub-id pub-id-type="doi">10.1126/science.aax2342</pub-id><pub-id pub-id-type="medline">31649194</pub-id></nlm-citation></ref><ref id="ref100"><label>100</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Srivarathan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bradford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shearkhani</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Bridging diagnostic safety and mental health: a systematic review highlighting inequities in autism spectrum disorder diagnosis</article-title><source>BMJ Qual Saf</source><year>2026</year><month>06</month><day>18</day><volume>35</volume><issue>7</issue><fpage>487</fpage><lpage>497</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2025-018723</pub-id><pub-id pub-id-type="medline">40854799</pub-id></nlm-citation></ref><ref id="ref101"><label>101</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lincoln</surname><given-names>A</given-names> </name><name name-style="western"><surname>Espejo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Johnson</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Limited literacy and psychiatric disorders among users of an urban safety-net hospital&#x2019;s mental health outpatient clinic</article-title><source>J Nerv Ment Dis</source><year>2008</year><month>09</month><volume>196</volume><issue>9</issue><fpage>687</fpage><lpage>693</lpage><pub-id pub-id-type="doi">10.1097/NMD.0b013e31817d0181</pub-id><pub-id pub-id-type="medline">18791430</pub-id></nlm-citation></ref><ref id="ref102"><label>102</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chowdhary</surname><given-names>N</given-names> </name><name name-style="western"><surname>Jotheeswaran</surname><given-names>AT</given-names> </name><name name-style="western"><surname>Nadkarni</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The methods and outcomes of cultural adaptations of psychological treatments for depressive disorders: a systematic review</article-title><source>Psychol Med</source><year>2014</year><month>04</month><volume>44</volume><issue>6</issue><fpage>1131</fpage><lpage>1146</lpage><pub-id pub-id-type="doi">10.1017/S0033291713001785</pub-id><pub-id pub-id-type="medline">23866176</pub-id></nlm-citation></ref><ref id="ref103"><label>103</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chong</surname><given-names>WW</given-names> </name><name name-style="western"><surname>Aslani</surname><given-names>P</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>TF</given-names> </name></person-group><article-title>Effectiveness of interventions to improve antidepressant medication adherence: a systematic review</article-title><source>Int J Clin Pract</source><year>2011</year><month>09</month><volume>65</volume><issue>9</issue><fpage>954</fpage><lpage>975</lpage><pub-id pub-id-type="doi">10.1111/j.1742-1241.2011.02746.x</pub-id><pub-id pub-id-type="medline">21849010</pub-id></nlm-citation></ref><ref id="ref104"><label>104</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osterberg</surname><given-names>L</given-names> </name><name name-style="western"><surname>Blaschke</surname><given-names>T</given-names> </name></person-group><article-title>Adherence to medication</article-title><source>N Engl J Med</source><year>2005</year><month>08</month><day>4</day><volume>353</volume><issue>5</issue><fpage>487</fpage><lpage>497</lpage><pub-id pub-id-type="doi">10.1056/NEJMra050100</pub-id><pub-id pub-id-type="medline">16079372</pub-id></nlm-citation></ref><ref id="ref105"><label>105</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kroenke</surname><given-names>K</given-names> </name><name name-style="western"><surname>Spitzer</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Williams</surname><given-names>JBW</given-names> </name></person-group><article-title>The PHQ-9: validity of a brief depression severity measure</article-title><source>J Gen Intern Med</source><year>2001</year><month>09</month><volume>16</volume><issue>9</issue><fpage>606</fpage><lpage>613</lpage><pub-id pub-id-type="doi">10.1046/j.1525-1497.2001.016009606.x</pub-id><pub-id pub-id-type="medline">11556941</pub-id></nlm-citation></ref><ref id="ref106"><label>106</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hamilton</surname><given-names>M</given-names> </name></person-group><article-title>A rating scale for depression</article-title><source>J Neurol Neurosurg Psychiatry</source><year>1960</year><month>02</month><volume>23</volume><issue>1</issue><fpage>56</fpage><lpage>62</lpage><pub-id pub-id-type="doi">10.1136/jnnp.23.1.56</pub-id><pub-id pub-id-type="medline">14399272</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Patient simulator.</p><media xlink:href="ai_v5i1e100772_app1.docx" xlink:title="DOCX File, 35 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Medical and behavioral profiles.</p><media xlink:href="ai_v5i1e100772_app2.docx" xlink:title="DOCX File, 145 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Retrieval performance.</p><media xlink:href="ai_v5i1e100772_app3.docx" xlink:title="DOCX File, 219 KB"/></supplementary-material></app-group></back></article>