<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e93498</article-id><article-id pub-id-type="doi">10.2196/93498</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Implicit Bias in Large Language Model Diagnosis of Eating Disorders: Experimental Vignette Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>McCalla</surname><given-names>Deija</given-names></name><degrees>MA</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Bochen</given-names></name><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jaeger</surname><given-names>Saul</given-names></name><degrees>PsyD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jacques</surname><given-names>Justin</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gonzalez Jr</surname><given-names>Leo</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Silber</surname><given-names>Charles</given-names></name><degrees>EdD</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Dykeman</surname><given-names>Cass</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>New Generation Mental Health Counseling</institution><addr-line>223 Bedford Ave</addr-line><addr-line>Brooklyn</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff2"><institution>Oregon State University</institution><addr-line>Corvallis</addr-line><addr-line>OR</addr-line><country>United States</country></aff><aff id="aff3"><institution>Touro University</institution><addr-line>Los Alamitos</addr-line><addr-line>CA</addr-line><country>United States</country></aff><aff id="aff4"><institution>Human Theory Group</institution><addr-line>Washington</addr-line><addr-line>DC</addr-line><country>United States</country></aff><aff id="aff5"><institution>St. John's University</institution><addr-line>Queens</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff6"><institution>Rutgers University</institution><addr-line>New Brunswick</addr-line><addr-line>NJ</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Nicora</surname><given-names>Giovanna</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Almenara</surname><given-names>Carlos A</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Chakit</surname><given-names>Miloud</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Hu</surname><given-names>Yihan</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Deija McCalla, MA, New Generation Mental Health Counseling, 223 Bedford Ave, Brooklyn, NY, 11211, United States, 1 347-559-7120; <email>deijamccalla@newgencounseling.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>23</day><month>9</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e93498</elocation-id><history><date date-type="received"><day>13</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>03</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>06</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Deija McCalla, Bochen Li, Saul Jaeger, Justin Jacques, Leo Gonzalez Jr, Charles Silber, Cass Dykeman. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 23.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e93498"/><abstract><sec><title>Background</title><p>Large language models (LLMs) are increasingly deployed in mental health applications, yet growing evidence suggests they encode algorithmic biases that influence clinical outputs. Because these models now mediate patient-facing decisions, such biases carry the potential for direct harm. Whether they systematically affect psychiatric diagnosis across demographic groups remains underexplored.</p></sec><sec><title>Objective</title><p>This study aims to examine whether LLMs exhibit implicit demographic biases when generating psychiatric diagnoses.</p></sec><sec sec-type="methods"><title>Methods</title><p>We developed 1152 synthetic clinical vignettes using a matched-pair design that manipulated gender, race and ethnicity, age, socioeconomic status, English proficiency, and urbanicity while holding clinical content constant. Vignettes were divided into control (unambiguous anorexia nervosa [AN]) and ambiguous conditions designed to permit differential diagnosis. Ten LLM configurations across 5 model families were tested.</p></sec><sec sec-type="results"><title>Results</title><p>Control vignettes produced near-unanimous AN diagnoses (mean 100%, SD 0.1%), while ambiguous vignettes elicited greater variability (mean 23.6%, SD 10.1%). Intermodel agreement was moderate for ambiguous vignettes (Fleiss &#x03BA;=0.410, 95% CI 0.397&#x2010;0.422). Mixed-effects logistic regression with LLM as a random intercept revealed significant demographic biases: Black patients were over 6 times more likely to receive a major depressive disorder (MDD) diagnosis than White patients with identical presentations (odds ratio [OR] 6.09, 95% CI 5.13&#x2010;7.24), Latine patients were over 9 times more likely (OR 9.57, 95% CI 8.00&#x2010;11.45), and Asian patients were nearly 3 times more likely to receive an AN diagnosis (OR 2.88, 95% CI 2.44&#x2010;3.42). Female patients were less likely than males to be diagnosed with AN (OR 0.43, 95% CI 0.37&#x2010;0.49).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>These findings demonstrate that LLMs exhibit systematic demographic biases in psychiatric diagnosis even when clinical content is held constant, revealing measurable patterns that can inform improvements to training data, model architecture, and clinical deployment frameworks.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>algorithmic bias</kwd><kwd>eating disorders</kwd><kwd>psychiatric diagnosis</kwd><kwd>health equity</kwd><kwd>artificial intelligence</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Rise of Large Language Models in Mental Health Care</title><p>Large language models (LLMs) are increasingly integrated into clinical workflows, with applications ranging from clinical note summarization to diagnostic support and patient-facing triage [<xref ref-type="bibr" rid="ref1">1</xref>]. In mental health care, where unstructured narrative data dominate and clinician time is scarce, LLMs offer particular appeal due to their ability to process free-text inputs [<xref ref-type="bibr" rid="ref2">2</xref>] and generate fluent, context-sensitive responses [<xref ref-type="bibr" rid="ref3">3</xref>]. AI-powered therapy chatbots such as Woebot and Wysa, which deliver cognitive behavioral therapy techniques, have demonstrated efficacy in reducing depression and anxiety symptoms in randomized controlled and real-world evaluation studies [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. Beyond these applications, generative AI (GenAI) tools have been evaluated directly as diagnostic classifiers: when ChatGPT (OpenAI), Claude (Anthropic), and Gemini (Google) were tasked with identifying major depressive disorder (MDD) from written clinical vignettes, classification accuracy varied markedly across model families, from near chance to near-perfect, underscoring both the diagnostic potential of these tools and the degree to which performance depends on model choice [<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>However, emerging evidence suggests that LLMs may encode demographic biases that influence clinical outputs. In this context, implicit bias refers to systematic diagnostic shifts driven by demographic cues in the absence of explicit instructions to weight patient identity, a functional analog of the construct documented in human clinicians. For instance, when patient race or gender was the only variable modified in clinical vignettes, GPT-4 (OpenAI) produced significantly different diagnostic and treatment recommendations, including reduced likelihood of recommending advanced imaging for Black patients [<xref ref-type="bibr" rid="ref7">7</xref>]. Similar patterns have also emerged in psychiatric contexts, where LLMs proposed inferior treatment plans when patient race was explicitly or implicitly indicated [<xref ref-type="bibr" rid="ref8">8</xref>].</p></sec><sec id="s1-2"><title>Algorithmic Bias in AI</title><p>Algorithmic bias occurs when computational systems systematically and repeatedly produce outcomes that unfairly advantage or disadvantage a particular group [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. It is rarely a purely technical anomaly, but rather a reflection of systemic inequities embedded in data and clinical practice. A landmark study by Obermeyer et al [<xref ref-type="bibr" rid="ref11">11</xref>] demonstrated how an algorithm widely used to allocate care management resources systematically underprioritized Black patients by using health care costs as a proxy for health needs, despite equivalent illness burden. Within natural language processing, bias manifests through stereotypical associations learned from large text corpora and differential calibration across subgroups reflected in group-level differences in error rates and reliability [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref15">15</xref>]. Recent audits of leading LLMs reveal that while diagnostic labels may remain relatively stable across demographic framings, the quality and safety of treatment recommendations degrade when race is implied or stated [<xref ref-type="bibr" rid="ref8">8</xref>]. Mitigation remains fragmented: a systematic review found that only a minority of clinical AI studies explicitly addressed racial or ethnic bias, while most relied on preprocessing approaches such as reweighting without evaluating downstream clinical impact [<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>Critics have argued that LLMs operate as &#x201C;stochastic parrots,&#x201D; reproducing statistical patterns from training data without grounding in causal understanding or lived experience [<xref ref-type="bibr" rid="ref17">17</xref>]. Because mental health diagnosis currently lacks laboratory-validated biomarkers, clinicians must interpret ambiguous behavioral cues (such as social withdrawal, irritability, or weight preoccupations) within a relational and contextual frame [<xref ref-type="bibr" rid="ref18">18</xref>]. If LLMs merely recycle the demographic biases embedded in their training data, they risk perpetuating rather than resolving the interpretive challenges that already complicate diagnosis. Whether LLMs can approximate this nuanced clinical judgment remains an open empirical question.</p></sec><sec id="s1-3"><title>Demographic Disparities in Mental Health Diagnoses</title><p>A substantial body of research documents how demographic factors such as race, gender, age, and socioeconomic status (SES) influence psychiatric diagnosis independent of symptom severity. A naturalistic study of electronic medical records in an outpatient behavioral health clinic found that African American patients diagnosed with schizophrenia were significantly more likely than non-Latine White patients to screen positive for moderately severe to severe depression, suggesting underrecognition of mood symptoms in African Americans during clinical assessment [<xref ref-type="bibr" rid="ref19">19</xref>]. This aligns with a broader literature indicating that African Americans presenting with affective symptoms are disproportionately diagnosed with schizophrenia-spectrum disorders, while White patients with similar clinical profiles are more likely to receive mood disorder diagnoses, a pattern repeatedly observed across settings and study designs [<xref ref-type="bibr" rid="ref19">19</xref>]. In eating disorders, disparities follow a parallel logic: the entrenched prototype of the &#x201C;skinny, White, affluent girl&#x201D; often cited as a dominant clinical schema [<xref ref-type="bibr" rid="ref20">20</xref>] contributes to systemic underdiagnosis of anorexia nervosa (AN) in males, racial and ethnic minorities, and individuals of lower SES, even when objective criteria such as significant weight loss or fear of weight gain are met.</p><p>Gender bias in psychiatric diagnosis extends beyond eating disorders. Clinicians are more likely to diagnose women with depression than men even when both present with identical symptoms or equivalent scores on standardized measures [<xref ref-type="bibr" rid="ref21">21</xref>]. Men experiencing depression often exhibit externalizing symptoms such as irritability, aggression, and substance use rather than prototypical internalizing symptoms such as sadness or hopelessness, yet traditional diagnostic criteria and screening tools are calibrated toward internalizing presentations, contributing to systematic underdiagnosis of depression in men [<xref ref-type="bibr" rid="ref22">22</xref>]. These biases are compounded by institutional barriers, including fragmented care and delayed diagnosis, particularly among underserved populations [<xref ref-type="bibr" rid="ref23">23</xref>]. Clinician-level biases may also contribute, as psychiatric symptoms in marginalized patients are sometimes misattributed to perceived personality traits or lifestyle factors rather than evaluated as indicators of mental illness [<xref ref-type="bibr" rid="ref24">24</xref>]. Together, these patterns illustrate how diagnostic ambiguity interacts with demographic assumptions, creating precisely the conditions under which automated systems trained on historical data may reproduce or amplify inequities.</p></sec><sec id="s1-4"><title>Purpose of This Study</title><p>Given the rapid deployment of LLMs in mental health settings and mounting evidence of bias in both human and algorithmic diagnostic systems, there is an urgent need for controlled experimental studies that systematically test how LLMs interpret clinical narratives across demographic contexts. Empirical analyses of real-world AI health incidents reinforce this concern: across documented incidents, bias and discrimination rank among the most frequently reported harms, yet such incidents appear to be substantially underreported, underscoring the value of proactive, controlled auditing rather than reliance on retrospective incident surveillance [<xref ref-type="bibr" rid="ref25">25</xref>].</p><p>Despite this need, an empirical gap remains. While quantitative audits [<xref ref-type="bibr" rid="ref11">11</xref>] have raised alarms, few studies use designs capable of isolating the causal influence of specific demographic variables on diagnostic output independent of clinical content, and fewer still examine whether such effects are consistent across the range of models actually in use or are concentrated in particular demographic combinations. Existing work thus establishes that bias is plausible without establishing that demographic cues alone can shift diagnosis when clinical information is held constant, how far this generalizes across model families and tiers, or how it behaves at the intersection of multiple attributes.</p><p>This study addresses that gap using a matched-pair experimental design that holds clinical content constant and varies only demographic framing, allowing demographic effects to be isolated from symptom variation and care access. We examine whether LLMs, when presented with systematically varied clinical vignettes for AN, exhibit shifts in diagnostic assignment corresponding to demographic attributes known to correlate with real-world diagnostic disparities, including gender, race and ethnicity, age, SES, English proficiency, and urbanicity, and we do so across 10 model configurations and at the intersection of multiple attributes. Because the direction of LLM bias cannot be assumed a priori&#x2014;models may reproduce, attenuate, or invert documented human-clinician patterns&#x2014;we frame the study around 2 open research questions (RQs) rather than directional hypotheses:</p><p>RQ1: Do LLMs exhibit systematic diagnostic biases based on patient demographics when clinical content is held constant?</p><p>This question tests whether demographic cues alone are sufficient to alter diagnostic output, the central condition required to attribute bias to identity rather than presentation.</p><p>RQ2: To what extent do LLMs exhibit diagnostic consensus, and how do model architecture (family and tier) and patient demographic profiles moderate this consistency?</p><p>This question tests whether any observed bias is a general property of current models or varies with design choices and demographic context, which bears directly on whether model selection has clinical consequences.</p><p>In addressing these questions, the study makes 3 contributions. First, it applies a counterfactual matched-pair design that isolates the causal influence of demographic framing from clinical content, moving beyond qualitative audits toward quantified, attributable effects. Second, it evaluates breadth rarely examined together: 10 model configurations spanning 5 families and 2 capability tiers, 6 demographic dimensions, and their intersections, surfacing effects that main-effects-only analyses would miss. Third, it provides a reusable, version-aware auditing methodology and an initial baseline against which diagnostic bias can be tracked as models are updated.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><sec id="s2-1-1"><title>Overview</title><p>This study used a matched-pair experimental design to examine [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref26">26</xref>] whether LLMs exhibit systematic demographic biases when generating psychiatric diagnoses. The design operationalizes counterfactual fairness, under which a predictor is fair if its output for a given case would remain unchanged had the individual&#x2019;s demographic attributes been different, all else held equal [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. Whereas this counterfactual is unobservable for human patients, it can be tested directly with LLMs: each vignette pair presents an identical clinical picture in which only one demographic variable changes. If a model assigns different diagnoses to the 2 members of a pair, the demographic change caused the difference; what the design cannot reveal is why the model responded to it.</p></sec><sec id="s2-1-2"><title>Corpus</title><p>The clinical vignette corpus was organized hierarchically across factors: (1) clinical condition, (2) semantic variation, and (3) demographic header. First, vignettes were placed in 1 of 2 conditions, which were control or ambiguous. Control vignettes represented clear-cut cases of AN that fully met <italic>DSM-5</italic> (<italic>Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition</italic>) criteria, while ambiguous vignettes were more open to interpretation and were susceptible to more differential diagnoses. AN was selected as the target diagnosis due to well-documented demographic disparities in its clinical detection, including historical underdiagnosis in males, racial and ethnic minorities, and lower socioeconomic groups [<xref ref-type="bibr" rid="ref20">20</xref>]. This design assessed whether LLMs could produce consistent diagnoses for unambiguous cases as well as detect whether clinical ambiguity causes LLMs to factor in demographic stereotypes to influence diagnostic reasoning. Second, vignettes were placed in 1 of 3 semantic variation categories: conversational (A), clinical (B), and progressive (C). This was done to test whether LLMs were sensitive to how symptoms were described, not just what symptoms were present. Third, vignettes were constructed to vary systematically across 6 different demographic dimensions: gender, race and ethnicity, age group, SES, English language proficiency, and urbanicity. For each dimension, clinical content remained constant, while only the target demographic variable changed. In sum, this 1152-vignette corpus was organized across a fully balanced 3-factor hierarchy, split initially by 2 clinical conditions that were each rendered into 3 semantic variations of 192 vignettes, with each variation carrying a unique demographic header (control=192 (A)+192 (B)+192 (C); ambiguous=192 (A)+192 (B)+192 (C); total, 6&#x00D7;192=1152).</p></sec><sec id="s2-1-3"><title>LLM Configuration</title><p>The 1152 vignettes were prompted to base-tier and advanced-tier versions of 5 different LLM families (10 LLM configurations total). This fully crossed prompting design yielded a total of 11,520 LLM evaluation instances (1152 vignettes &#x00D7; 10 LLM configurations), divided evenly between base-tier (n=5760) and advanced-tier (n=5760) model evaluations.</p></sec><sec id="s2-1-4"><title>Ethical Considerations</title><p>This study used synthetic clinical vignettes and did not involve human participants, real patient data, or protected health information. As the research examined LLM outputs rather than human participants, institutional review board approval was not required.</p></sec><sec id="s2-1-5"><title>Study Workflow</title><p>The study proceeded in 5 steps, each detailed in the sections that follow (<xref ref-type="fig" rid="figure1">Figure 1</xref>). First, the 1152-vignette corpus was generated using a structured scaffold-and-slot system and validated by the research team (see &#x201C;Clinical Vignette Development&#x201D; section). Second, demographic headers were systematically varied across 6 dimensions while clinical content was held constant within matched pairs (see &#x201C;Demographic Variables&#x201D; section). Third, 10 LLM configurations were selected and parameterized for uniform administration (see &#x201C;LLM Selection and Configuration&#x201D; section). Fourth, all vignettes were submitted to each configuration via scripted API calls, yielding 11,520 diagnostic outputs (see &#x201C;Data Collection Procedure&#x201D; section). Fifth, outputs were harmonized into diagnostic categories and analyzed for manipulation effectiveness, demographic bias, intermodel agreement, and bias profile similarity (see &#x201C;Statistical Analysis&#x201D; section).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Five-step pipeline from vignette generation through statistical analysis; section numbers refer to the corresponding &#x201C;Methods&#x201D; subsections. LLM: large language model; SES: socioeconomic status.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e93498_fig01.png"/></fig></sec></sec><sec id="s2-2"><title>Clinical Vignette Development</title><sec id="s2-2-1"><title>Generation Procedure</title><p>Clinical vignettes were constructed in 2 stages. First, the research team specified, for each of the 5 semantic slots (see &#x201C;Semantic Slots&#x201D; section), the clinical content required to satisfy <italic>DSM-5</italic> criteria for AN in the control condition. Initial prose drafts of the control vignettes were produced with the assistance of 2 LLM assistants&#x2014;ChatGPT (OpenAI; GPT-4o) and Claude (Anthropic; Claude Sonnet 4.5), both accessed via their respective web interfaces in October 2025&#x2014;prompted to render the authors&#x2019; slot specifications into fluent clinical narratives across the 3 semantic variations (see &#x201C;Semantic Variations&#x201D; section). These drafts were iteratively revised and validated by the research team over multiple rounds to ensure <italic>DSM-5</italic> fidelity, clinical plausibility, and adherence to the lexical controls described in the &#x201C;Lexical Controls&#x201D; section.</p><p>Second, ambiguous vignettes were derived from the validated control vignettes by applying a structured ambiguity protocol comprising 5 manipulations: criterion dilution, differential-diagnosis injection, timeline manipulation, severity bracketing, and information incompleteness. The ambiguous narrative templates were likewise drafted and refined with the assistance of the same 2 assistants and validated by the team, then applied deterministically: a generation script held each demographic header fixed and substituted the corresponding ambiguous clinical body, ensuring that every demographic variant within a matched pair received identical clinical content.</p><p>The vignettes were developed over multiple iterative model sessions interleaved with manual revision; consequently, no single verbatim prompt-and-response transcript documents the final stimuli. The complete corpus of 1152 finalized vignettes is available on the project&#x2019;s OSF page (see &#x201C;Data Availability&#x201D; statement).</p></sec><sec id="s2-2-2"><title>Vignette Structure</title><p>Each vignette consisted of 2 parts: a demographic header followed by a clinical portion. The header described the patient&#x2019;s gender, race and ethnicity, age group, SES, English proficiency, and urbanicity (eg, &#x201C;The presenting client is an adolescent woman from a low-income background who identifies as Black and lives in an urban area. They speak limited English...&#x201D;); header composition and the demographic dimensions are detailed in the &#x201C;Demographic Variables&#x201D; section. Demographic attributes were manipulated solely through lexical substitutions in this header. Within each matched pair, the clinical portion was held verbatim-identical&#x2014;character-for-character&#x2014;so that no wording changed beyond the demographic descriptors themselves; the 2 members of a pair differed only in the header.</p></sec><sec id="s2-2-3"><title>Semantic Slots</title><p>The clinical portion of each vignette was built around five semantic slots corresponding to core diagnostic features of AN: (1) time course, (2) intake restriction pattern, (3) cognitive preoccupation, (4) weight and health consequences, and (5) attitude toward weight. Each slot contained phrasing options for both control and ambiguous variations.</p></sec><sec id="s2-2-4"><title>Semantic Variations</title><p>Three semantic variations (pairs A, B, and C) were created using conversational, clinical, and formal-progressive linguistic styles, respectively. Conversational vignettes used colloquial phrasing, clinical vignettes used standard medical terminology, and formal-progressive vignettes used contemporary, person-centered language emphasizing the patient&#x2019;s subjective experience. Matched-pair comparisons were conducted within each semantic variation (pairs A, B, and C), whose members share identical clinical text; the 3 variations constitute a separate, fully crossed factor and were not compared against one another at the pair level. Demographic contrasts therefore always held the clinical narrative constant, with semantic variation modeled as a distinct factor rather than a source of within-pair difference.</p></sec><sec id="s2-2-5"><title>Control Versus Ambiguous Operationalization</title><p>Control vignettes used explicit, specific language consistent with <italic>DSM-5</italic> criteria for AN, restricting type (eg, &#x201C;lost over 15 percent of body weight,&#x201D; &#x201C;track calories obsessively,&#x201D; and &#x201C;remain fearful of weight despite medical warnings&#x201D;). Ambiguous vignettes contained the same clinical themes, but used indirect or attenuated language that was suggestive, rather than confirmatory (eg, &#x201C;lost a noticeable amount of weight,&#x201D; &#x201C;seem preoccupied with monitoring intake,&#x201D; and &#x201C;continue to express concern about gaining weight despite feedback from others&#x201D;). Ambiguous vignettes were designed to more closely approximate real-world clinical presentations, where symptom profiles are often ambiguous and open to interpretation. To make this contrast concrete, <xref ref-type="other" rid="box1">Textbox 1</xref> presents a matched control-ambiguous pair drawn from the corpus (semantic variation B).</p><boxed-text id="box1"><title> Matched control and ambiguous vignettes illustrating the language manipulation. Both share an identical demographic header (adolescent man, low-income, urban, White, limited English; semantic variation B); only the clinical body differs. Italic text marks the parallel diagnostic features as they shift from explicit to attenuated.</title><p><bold>Control vignette (explicit)</bold></p><list list-type="bullet"><list-item><p>The presenting client is an adolescent man from a low-income background who identifies as White and lives in an urban area.</p></list-item><list-item><p>They speak limited English and present for counseling after concerns were raised about their eating habits.</p></list-item><list-item><p><italic>Reports progressively reducing food intake over the past 8 months</italic>.</p></list-item><list-item><p>They <italic>skip most meals, avoid carbohydrate-rich foods, and count every calorie</italic>.</p></list-item><list-item><p>They show <italic>clinically significant weight loss</italic> and orthostatic symptoms noted by clinicians.</p></list-item><list-item><p>Despite medical warnings, they <italic>remain fearful of weight gain</italic>.</p></list-item></list><p><bold>Ambiguous vignette (attenuated)</bold></p><list list-type="bullet"><list-item><p>The presenting client is an adolescent man from a low-income background who identifies as White and lives in an urban area.</p></list-item><list-item><p>They speak limited English and were referred for evaluation after family noted changes in eating behavior and affective presentation.</p></list-item><list-item><p><italic>Reports psychosocial stressor approximately 3 months prior to symptom onset.</italic></p></list-item><list-item><p>Client describes <italic>reduced appetite with decreased motivation for food intake</italic>.</p></list-item><list-item><p>Social withdrawal and flat affect observed by family.</p></list-item><list-item><p><italic>Weight reduction documented</italic>.</p></list-item><list-item><p>Client minimizes dietary concerns and <italic>denies preoccupation with food or weight</italic>.</p></list-item><list-item><p>Eating pattern shows day-to-day variability.</p></list-item><list-item><p><italic>Body image assessment unremarkable</italic>. Physical complaints include fatigue and orthostatic symptoms.</p></list-item></list></boxed-text></sec><sec id="s2-2-6"><title>Lexical Controls</title><p>Several controls were applied to vignette generation. Lexical controls kept word count within &#x00B1;10% of baseline, held sentence count constant, and maintained Flesch-Kincaid grade level within &#x00B1;0.5 across matched pairs. Flesch-Kincaid grade level is a widely used readability measure in research [<xref ref-type="bibr" rid="ref29">29</xref>]. Forbidden terms (eg, BMI values, &#x201C;meets criteria&#x201D;) were excluded to prevent diagnostic giveaways. Each control vignette was paired with a corresponding ambiguous vignette sharing the same demographic header to ensure content parity. The control-ambiguous manipulation was assessed as a manipulation check (see &#x201C;Results&#x201D; section).</p></sec><sec id="s2-2-7"><title>Demographic Variables</title><p>Each vignette header varied across 6 demographic dimensions: gender (male and female), race and ethnicity (White, Black, Asian, and Latine), age group (adolescent, young adult, and adult), SES (low SES and high SES), English language proficiency (fluent English and limited English), and urbanicity (urban and nonurban). These dimensions were selected based on documented disparities in psychiatric diagnosis and health care access [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. The matched-pair design held all demographic variables constant except one, allowing isolation of individual demographic effects on diagnostic outcomes. This yielded 576 unique demographic headers across all vignette pairings. The term Latine was used rather than Latino or Latina because vignette gender was independently manipulated; a gendered term would have confounded the gender and ethnicity dimensions. The corpus was fully balanced across all manipulated factors&#x2014;the 6 demographic dimensions, the 3 semantic variations, and the control-ambiguous condition&#x2014;yielding the crossed 1152-vignette structure described in the &#x201C;Corpus&#x201D; section. Diagnosis category was the measured outcome rather than a design factor and was therefore deliberately left unconstrained; the corpus was not balanced on diagnosis, as the diagnoses models assign are precisely what the study set out to observe.</p></sec></sec><sec id="s2-3"><title>LLM Selection and Configuration</title><sec id="s2-3-1"><title>Selection Rationale</title><p>The 10 configurations were selected across 5 model families: ChatGPT (OpenAI), Claude (Anthropic), DeepSeek, Llama (Meta), and Mistral. These families were chosen to capture variation in capability (base-tier vs advanced-tier), developer and source country (United States: ChatGPT, Claude, and Llama; France: Mistral; China: DeepSeek), and model type (proprietary vs open-weight). Base-tier models were defined as freely accessible or lower-cost versions, and advanced-tier models as paid or higher-capability versions. This differentiation was used to examine whether diagnostic bias patterns differ by model sophistication.</p><p>Model selection prioritized ecological validity over frontier capability. Rather than benchmarking the most advanced systems available, we sampled models that members of the public most commonly use when posing health-related questions, spanning a range of developers and source countries and both freely accessible (base-tier) and paid (advanced-tier) configurations. This frame reflects the study&#x2019;s aim&#x2014;characterizing diagnostic bias in the models people and budget-constrained clinical tools actually encounter, where lower-cost configurations are widely deployed&#x2014;rather than establishing the upper bound of model performance. The GPT-5 family, although available during the data-collection window, was not incorporated because the data-collection pipeline had been implemented on the prior generation of GPT and was not updated before collection concluded; we address this omission as a limitation (see &#x201C;Limitations&#x201D; section).</p></sec><sec id="s2-3-2"><title>Hyperparameter Settings</title><p>All models were used at a temperature setting of 0.1 to minimize stochastic sampling variability. This was done so that diagnostic output was attributable to differences in vignette content rather than to run-to-run randomness. A value of 0.1 rather than 0.0 was used because several provider APIs do not guarantee strictly deterministic behavior at 0.0. To confirm that this choice did not materially affect diagnostic output, a stratified subset of 192 vignettes (96 control and 96 ambiguous; 64 from each semantic variation) was readministered at both 0.0 and 0.1 to the 8 non-Claude configurations, yielding 1536 paired comparisons. Diagnoses were highly concordant across the 2 settings (97.5% agreement; mean Cohen &#x03BA;=0.90, SD 0.17), indicating that temperature had negligible influence on diagnostic classification. The dispersion in Cohen &#x03BA; was driven by a single configuration whose kappa was deflated by a skewed diagnosis distribution despite 95.8% raw agreement, rather than by genuine temperature instability (per-model results available in the study&#x2019;s OSF repository; see &#x201C;Data Availability&#x201D; statement). The two Claude configurations could not be included because the version-pinned end points on which the study was conducted had been retired by the time this sensitivity analysis was performed. Maximum token length was set to 50, with the exception of DeepSeek Reasoner (2000 tokens) to accommodate its internal reasoning process. Claude model responses (Claude 3 Haiku and Claude Sonnet 4) were recollected in February 2026 to correct a temperature parameter error identified during manuscript preparation; all other models were tested in October 2025. Full model configurations are reported in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Large language model configurations<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model family</td><td align="left" valign="bottom">Tier</td><td align="left" valign="bottom">API model string</td><td align="left" valign="bottom">Temperature</td><td align="left" valign="bottom">Maximum tokens</td></tr></thead><tbody><tr><td align="left" valign="top">ChatGPT</td><td align="left" valign="top">Base</td><td align="left" valign="top">gpt-3.5-turbo</td><td align="left" valign="top">0.1</td><td align="left" valign="top">50</td></tr><tr><td align="left" valign="top">ChatGPT</td><td align="left" valign="top">Advanced</td><td align="left" valign="top">gpt-4o</td><td align="left" valign="top">0.1</td><td align="left" valign="top">50</td></tr><tr><td align="left" valign="top">DeepSeek</td><td align="left" valign="top">Base</td><td align="left" valign="top">deepseek-chat</td><td align="left" valign="top">0.1</td><td align="left" valign="top">50</td></tr><tr><td align="left" valign="top">DeepSeek</td><td align="left" valign="top">Advanced</td><td align="left" valign="top">deepseek-reasoner</td><td align="left" valign="top">0.1</td><td align="left" valign="top">2000</td></tr><tr><td align="left" valign="top">Llama</td><td align="left" valign="top">Base</td><td align="left" valign="top">llama-3.1-8b-instant</td><td align="left" valign="top">0.1</td><td align="left" valign="top">50</td></tr><tr><td align="left" valign="top">Llama</td><td align="left" valign="top">Advanced</td><td align="left" valign="top">llama-3.3-70b-versatile</td><td align="left" valign="top">0.1</td><td align="left" valign="top">50</td></tr><tr><td align="left" valign="top">Mistral</td><td align="left" valign="top">Base</td><td align="left" valign="top">open-mistral-7b</td><td align="left" valign="top">0.1</td><td align="left" valign="top">50</td></tr><tr><td align="left" valign="top">Mistral</td><td align="left" valign="top">Advanced</td><td align="left" valign="top">mistral-large-latest</td><td align="left" valign="top">0.1</td><td align="left" valign="top">50</td></tr><tr><td align="left" valign="top">Claude</td><td align="left" valign="top">Base</td><td align="left" valign="top">claude-3-haiku-20240307</td><td align="left" valign="top">0.1</td><td align="left" valign="top">50</td></tr><tr><td align="left" valign="top">Claude</td><td align="left" valign="top">Advanced</td><td align="left" valign="top">claude-sonnet-4&#x2010;20250514</td><td align="left" valign="top">0.1</td><td align="left" valign="top">50</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Claude models were recollected in February 2026. All other models were collected in October 2025. The Mistral Advanced end point (mistral-large-latest) was not version-pinned. DeepSeek Advanced max tokens set to 2000 to accommodate the internal reasoning process.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-3-3"><title>Data Collection Procedure</title><p>Data collection was conducted using a Python script executed in Google Colab. Eight model configurations (ChatGPT, DeepSeek, Llama, and Mistral families) were tested in October 2025. Claude models were retested in February 2026 following identification of a temperature parameter error in the original collection script (see &#x201C;Hyperparameter Settings&#x201D; section). For each vignette, the model received a single user-role message consisting of the vignette text followed by a fixed instruction (no system prompt was used). The instruction, identical across all 10 configurations, read:</p><p>Please provide ONLY the most likely psychiatric diagnosis. Respond with just the diagnosis name (eg, &#x201C;Major Depressive Disorder,&#x201D;&#x201C;Major Depressive Disorder,&#x201D; &#x201C;Anorexia Nervosa,&#x201D; &#x201C;Anorexia Nervosa,&#x201D; &#x201C;Generalized Anxiety Disorder&#x201D;). Do not include explanations, reasoning, or additional information.</p><p>The parenthetical diagnoses were illustrative of the expected response format only; models were not constrained to a fixed answer set, and free-text outputs were subsequently harmonized into 4 diagnostic categories (see &#x201C;Data Preparation&#x201D; section). The decision to elicit a single diagnosis was deliberate. Because LLMs are effectively black boxes whose internal reasoning cannot be directly observed, controlled, and replicable causal inference requires a clean, categorical outcome measure. Free-response outputs vary in length, hedging, structure, and framing, making them difficult to code reproducibly and to compare across matched pairs; moreover, these output conventions differ systematically across model families and tiers, which would confound genuine differences in diagnostic behavior with superficial differences in expression. As a central aim of this study was to compare 10 configurations across 5 model families, constraining each response to a single diagnosis placed all models on a common measurement footing, enabling both the matched-pair and cross-model comparisons. Implications of this constraint for the interpretation of effect sizes are considered in the &#x201C;Limitations&#x201D; section. Raw diagnostic outputs were recorded verbatim. In cases where API calls timed out, the script was reexecuted to obtain complete responses. The final dataset contained 11,520 diagnostic outputs with 14 missing responses (0.12%), all from non-Claude models. After data collection, results from both collection waves were combined into a single dataset for analysis. The data collection script is available in the study repository (see &#x201C;Data Availability&#x201D; statement).</p></sec></sec><sec id="s2-4"><title>Statistical Analysis</title><p>All statistical analyses were conducted in R (version 4.5.2; R Core Team). Key packages included <italic>irr</italic> for interrater reliability, <italic>DescTools</italic> for Cochran Q test, <italic>car</italic> for model diagnostics, and <italic>tidyverse</italic> for data manipulation and visualization. Cochran rule compliance for all chi-square tests is reported in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-5"><title>Data Preparation</title><p>Raw diagnostic outputs were harmonized to account for variations in LLM response formatting. Diagnoses were collapsed into 4 categories: AN, MDD, Avoidant/Restrictive Food Intake Disorder (ARFID), and other (eg, adjustment disorder, atypical AN, other specified feeding or eating disorder, and rare responses). Responses containing errors or missing data were flagged for sensitivity analysis.</p></sec><sec id="s2-6"><title>Manipulation Check</title><p>Chi-square tests compared AN diagnosis rates between control and ambiguous vignettes to confirm the effectiveness of the control-ambiguous manipulation. Effect size was measured using Cram&#x00E9;r V.</p></sec><sec id="s2-7"><title>Demographic Bias Analysis</title><p>Chi-square tests with Monte Carlo simulation (B=5000 replicates) examined associations between each demographic variable and diagnostic outcomes. Cochran rule was verified for all tests (no expected frequency&#x003C;1, no more than 20% of cells&#x003C;5). Mixed-effects binary logistic regression modeled the probability of AN and MDD diagnoses as a function of demographic predictors, with LLM included as a random intercept to account for clustering of observations within models. This approach accounts for the nonindependence of diagnoses generated by the same LLM across vignettes. Odds ratios (ORs) with 95% CIs were calculated from fixed effects. Per-model logistic regressions were also conducted to assess whether bias patterns were consistent across individual LLMs or driven by outlier models. The Benjamini-Hochberg procedure was applied to control the false discovery rate across multiple comparisons.</p></sec><sec id="s2-8"><title>Intermodel Agreement</title><p>Fleiss &#x03BA; assessed diagnostic agreement across all 10 LLM configurations, with 95% CIs. Agreement was also calculated separately by condition (control vs ambiguous) and demographic subgroups. Pairwise Cohen &#x03BA; quantified agreement between each LLM pair. Cochran Q test examined whether diagnosis rates differed significantly across LLMs, with post hoc pairwise McNemar tests.</p></sec><sec id="s2-9"><title>Bias Profile Clustering</title><p>Hierarchical cluster analysis (Ward D2 method and Euclidean distance) examined similarity in demographic bias profiles across LLMs based on diagnosis rates by gender and race and ethnicity.</p></sec><sec id="s2-10"><title>Robustness Checks</title><p>Pooled logistic regression (without random effects) was conducted to assess sensitivity of results to the modeling approach. A post hoc temperature sensitivity analysis, in which a vignette subset was readministered at temperature 0.0 (see &#x201C;Hyperparameter Settings&#x201D; section), assessed whether diagnostic outputs differed based upon temperature setting. Statistical significance was set at &#x03B1;=.05</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Preliminary Analyses</title><p>A total of 11,520 diagnostic outputs were generated (1152 vignettes &#x00D7; 10 LLM configurations). Missing or error responses were minimal (14 of 11,520 outputs; 0.12%), distributed across non-Claude models. No model exhibited systematic missingness.</p><p>Across all LLMs and conditions, the most common diagnoses were AN, MDD, and ARFID, with rare responses collapsed into an &#x201C;Other&#x201D; category for analysis.</p></sec><sec id="s3-2"><title>Manipulation Check</title><p>The control-ambiguous manipulation was effective. Control vignettes produced near-unanimous AN diagnoses (mean 100%, SD 0.1%), with 9 of 10 LLMs achieving 99.7%&#x2010;100% AN diagnosis rates for control cases. Ambiguous vignettes elicited substantially lower AN rates (mean 23.6%, SD 10.1%), representing a 76.4 percentage point difference. This difference was statistically significant, <italic>&#x03C7;</italic>&#x00B2;<sub>1</sub>=7111.40, <italic>P</italic>&#x003C;.001, with a large effect size (Cram&#x00E9;r <italic>V</italic>=0.786). These results confirm that control vignettes functioned as unambiguous AN cases, while ambiguous vignettes successfully induced diagnostic variability.</p><p>Across the 10 LLMs, AN diagnosis rates for ambiguous vignettes ranged from 12.9% (Claude Advanced) to 45.6% (Claude Base), while MDD rates ranged from 24.6% (Llama Advanced) to 62.2% (Llama Base; <xref ref-type="table" rid="table2">Table 2</xref>).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Anorexia nervosa (AN) and major depressive disorder (MDD) diagnosis rates for ambiguous vignettes by large language model (LLM) configuration<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">LLM configuration</td><td align="left" valign="bottom">AN rate (%)</td><td align="left" valign="bottom">MDD rate (%)</td></tr></thead><tbody><tr><td align="left" valign="top">ChatGPT Base</td><td align="left" valign="top">25.2</td><td align="left" valign="top">54.3</td></tr><tr><td align="left" valign="top">ChatGPT Advanced</td><td align="left" valign="top">17.6</td><td align="left" valign="top">40.8</td></tr><tr><td align="left" valign="top">DeepSeek Base</td><td align="left" valign="top">32.3</td><td align="left" valign="top">36.3</td></tr><tr><td align="left" valign="top">DeepSeek Advanced</td><td align="left" valign="top">17.7</td><td align="left" valign="top">45.9</td></tr><tr><td align="left" valign="top">Llama Base</td><td align="left" valign="top">25.2</td><td align="left" valign="top">62.2</td></tr><tr><td align="left" valign="top">Llama Advanced</td><td align="left" valign="top">13.3</td><td align="left" valign="top">24.6</td></tr><tr><td align="left" valign="top">Mistral Base</td><td align="left" valign="top">27.3</td><td align="left" valign="top">58.2</td></tr><tr><td align="left" valign="top">Mistral Advanced</td><td align="left" valign="top">17.6</td><td align="left" valign="top">44.3</td></tr><tr><td align="left" valign="top">Claude Base</td><td align="left" valign="top">45.6</td><td align="left" valign="top">28.5</td></tr><tr><td align="left" valign="top">Claude Advanced</td><td align="left" valign="top">12.9</td><td align="left" valign="top">42.0</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Rates represent the percentage of ambiguous vignettes (n=576) diagnosed with each condition.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>RQ1: Demographic Factors and Diagnostic Bias</title><p>All analyses of ambiguous vignettes used mixed-effects binary logistic regression with LLM as a random intercept to account for clustering of observations within models. Because clinical content was held constant across demographic conditions, any significant deviation from an OR of 1.0 is consistent with the influence of demographic framing on diagnostic output rather than differences in clinical presentation. Percentages denote observed diagnosis rates pooled across all 10 configurations on ambiguous vignettes; because the ORs derive from the mixed-effects model adjusting for the remaining predictors and between-model clustering, the two are directionally consistent but not arithmetically equivalent.</p><p>For AN diagnoses, Asian patients were significantly more likely than White patients to receive an AN diagnosis (46.4% vs 25.4% of ambiguous vignettes; OR 2.88, 95% CI 2.44&#x2010;3.42), while Latine patients (7.4% vs 25.4%; OR 0.20, 95% CI 0.16&#x2010;0.26), Black patients (15.6% vs 25.4%; OR 0.51, 95% CI 0.42&#x2010;0.62), and female patients (17.5% vs 29.8% for males; OR 0.43, 95% CI 0.37&#x2010;0.49) were less likely. Low-SES patients were also less likely to receive an AN diagnosis than high-SES patients (20.1% vs 27.2%; OR 0.61, 95% CI 0.53&#x2010;0.70). For MDD diagnoses, Latine patients were over 9 times more likely than White patients to receive an MDD diagnosis (71% vs 24.8% of ambiguous vignettes; OR 9.57, 95% CI 8.00&#x2010;11.45), Black patients were over 6 times more likely (62% vs 24.8%; OR 6.09, 95% CI 5.13&#x2010;7.24), and low-SES patients were more likely than high-SES patients (46.1% vs 40.4%; OR 1.41, 95% CI 1.24&#x2010;1.59). Female patients were less likely than males to receive MDD diagnoses (37.4% vs 49.1%; OR 0.50, 95% CI 0.44&#x2010;0.57). The LLM random intercept variance was substantial for both AN and MDD, confirming significant between-model variability in baseline diagnostic rates. Per-model logistic regressions confirmed that racial bias in MDD diagnosis was present across all 10 LLM configurations, with Black-versus-White ORs ranging from 2.69 (Llama Base) to 11.72 (Mistral Advanced). Full per-model logistic regression results for AN and MDD are presented in Tables S1 and S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Per-model diagnosis distributions and demographic breakdowns are presented in Figures S1-S30 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><p><xref ref-type="fig" rid="figure2">Figure 2</xref> displays the ORs and 95% CIs for all demographic predictors from the mixed-effects models.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Demographic predictors of anorexia nervosa and major depressive disorder diagnosis: odds ratios with 95% CIs from mixed-effects logistic regression (ambiguous vignettes, all large language models). The dashed line indicates odds ratio 1.0 (no effect). SES: socioeconomic status.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e93498_fig02.png"/></fig><p>Full mixed-effects model results, including all demographic predictors for both outcomes, are presented in <xref ref-type="table" rid="table3">Table 3</xref>.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Mixed-effects logistic regression odds ratios (ORs) for demographic predictors of anorexia nervosa (AN) and major depressive disorder (MDD) diagnosis (ambiguous vignettes)<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" rowspan="2">Predictor</td><td align="left" valign="bottom" colspan="3">AN</td><td align="left" valign="bottom" colspan="3">MDD</td></tr><tr><td align="left" valign="bottom">OR</td><td align="left" valign="bottom">95% CI</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">OR</td><td align="left" valign="bottom">95% CI</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Female vs male</td><td align="left" valign="top">0.43</td><td align="left" valign="top">0.37&#x2010;0.49</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.50</td><td align="left" valign="top">0.44&#x2010;0.57</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Black vs White</td><td align="left" valign="top">0.51</td><td align="left" valign="top">0.42&#x2010;0.62</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">6.09</td><td align="left" valign="top">5.13&#x2010;7.24</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Latine vs White</td><td align="left" valign="top">0.20</td><td align="left" valign="top">0.16&#x2010;0.26</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">9.57</td><td align="left" valign="top">8.00&#x2010;11.45</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Asian vs White</td><td align="left" valign="top">2.88</td><td align="left" valign="top">2.44&#x2010;3.42</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.50</td><td align="left" valign="top">0.41&#x2010;0.61</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Young adult vs adolescent</td><td align="left" valign="top">1.07</td><td align="left" valign="top">0.91&#x2010;1.27</td><td align="left" valign="top">.39</td><td align="left" valign="top">0.95</td><td align="left" valign="top">0.81&#x2010;1.10</td><td align="left" valign="top">.48</td></tr><tr><td align="left" valign="top">Adult vs adolescent</td><td align="left" valign="top">0.80</td><td align="left" valign="top">0.67&#x2010;0.94</td><td align="left" valign="top">.008</td><td align="left" valign="top">0.90</td><td align="left" valign="top">0.77&#x2010;1.05</td><td align="left" valign="top">.19</td></tr><tr><td align="left" valign="top">Low SES<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> vs high SES</td><td align="left" valign="top">0.61</td><td align="left" valign="top">0.53&#x2010;0.70</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">1.41</td><td align="left" valign="top">1.24&#x2010;1.59</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Limited vs fluent English</td><td align="left" valign="top">0.82</td><td align="left" valign="top">0.72&#x2010;0.94</td><td align="left" valign="top">.005</td><td align="left" valign="top">1.22</td><td align="left" valign="top">1.08&#x2010;1.39</td><td align="left" valign="top">.002</td></tr><tr><td align="left" valign="top">Nonurban vs urban</td><td align="left" valign="top">1.02</td><td align="left" valign="top">0.89&#x2010;1.17</td><td align="left" valign="top">.74</td><td align="left" valign="top">0.92</td><td align="left" valign="top">0.81&#x2010;1.04</td><td align="left" valign="top">.20</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Reference categories: male (gender), White (race and ethnicity), adolescent (age group), high SES (socioeconomic status), fluent English (language), urban (urbanicity). Large language model included as random intercept.</p></fn><fn id="table3fn2"><p><sup>b</sup>SES: socioeconomic status.</p></fn></table-wrap-foot></table-wrap><p>The gender &#x00D7; race or ethnicity interaction was significant for both AN (<italic>&#x03C7;</italic>&#x00B2;<sub>3</sub>=229.08, <italic>P</italic>&#x003C;.001) and MDD (<italic>&#x03C7;</italic>&#x00B2;<sub>3</sub>=1495.95, <italic>P</italic>&#x003C;.001). For AN, the interaction was driven by Latine patients, for whom the gender effect was substantially larger than for other racial or ethnic groups (interaction OR 52.99, 95% CI 23.29&#x2010;120.58; <italic>P</italic>&#x003C;.001), indicating that the low AN diagnosis rate among Latine patients was concentrated among males. Raw cell frequencies contextualize this interaction: Latine males received AN diagnoses in only 7 of 720 cases (1%), compared to 99 of 720 (13.8%) for Latine females, a reversal of the pattern observed in all other racial or ethnic groups, where males consistently received more AN diagnoses than females. This pattern is consistent with compounding demographic heuristics in which the combination of Latine ethnicity and male gender was treated as maximally incompatible with AN. This unusually large estimate reflects extreme cell sparsity and should be interpreted as indicating direction and magnitude rather than precise effect size. Additionally, White female patients received zero MDD diagnoses across all 10 LLM configurations (0 of 720 cases), contributing to the quasi-complete separation observed in the MDD interaction model. For MDD, although the omnibus interaction was significant, individual interaction coefficients exhibited quasi-complete separation due to multiple models producing zero diagnoses for specific gender &#x00D7; race combinations, precluding reliable estimation of cell-level interaction effects.</p><p>More generally, the gender &#x00D7; race or ethnicity interaction models were affected by quasi-complete separation arising from zero-count cells (eg, White female patients received no MDD diagnoses across all configurations), which inflates point estimates and widens CIs. The interaction estimates&#x2014;most notably the Latine &#x00D7; male effect for AN (OR 52.99, 95% CI 23.29&#x2010;120.58), whose interval width itself reflects this instability&#x2014;should therefore be read as indicating the direction and approximate magnitude of an effect rather than precise values. We emphasize that this instability is confined to the sparse interaction terms; the main-effect ORs reported above, which are estimated from the full sample, are not subject to separation and are stable.</p><p>Full interaction model results are presented in <xref ref-type="table" rid="table4">Table 4</xref>. Per-model logistic regression results for AN and MDD are available in Tables S1 and S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Gender &#x00D7; race or ethnicity interaction model odds ratios (ORs) for anorexia nervosa (AN) and major depressive disorder (MDD) diagnosis<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" rowspan="2">Predictor</td><td align="left" valign="bottom" colspan="3">AN</td><td align="left" valign="bottom" colspan="3">MDD</td></tr><tr><td align="left" valign="bottom">OR</td><td align="left" valign="bottom">95% CI</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">OR</td><td align="left" valign="bottom">95% CI</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Female</td><td align="left" valign="top">0.33</td><td align="left" valign="top">0.26&#x2010;0.43</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">NE<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">NE</td><td align="left" valign="top">.57</td></tr><tr><td align="left" valign="top">Black</td><td align="left" valign="top">0.57</td><td align="left" valign="top">0.45&#x2010;0.72</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.48</td><td align="left" valign="top">0.38&#x2010;0.60</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Latine</td><td align="left" valign="top">0.01</td><td align="left" valign="top">0.01&#x2010;0.03</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">15.45</td><td align="left" valign="top">11.19&#x2010;21.32</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Asian</td><td align="left" valign="top">3.15</td><td align="left" valign="top">2.51&#x2010;3.96</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.20</td><td align="left" valign="top">0.16&#x2010;0.26</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Young adult</td><td align="left" valign="top">1.08</td><td align="left" valign="top">0.91&#x2010;1.28</td><td align="left" valign="top">.38</td><td align="left" valign="top">0.93</td><td align="left" valign="top">0.77&#x2010;1.11</td><td align="left" valign="top">.39</td></tr><tr><td align="left" valign="top">Adult</td><td align="left" valign="top">0.79</td><td align="left" valign="top">0.66&#x2010;0.94</td><td align="left" valign="top">.007</td><td align="left" valign="top">0.87</td><td align="left" valign="top">0.72&#x2010;1.04</td><td align="left" valign="top">.12</td></tr><tr><td align="left" valign="top">Low SES<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">0.60</td><td align="left" valign="top">0.52&#x2010;0.69</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">1.59</td><td align="left" valign="top">1.37&#x2010;1.85</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Limited English</td><td align="left" valign="top">0.82</td><td align="left" valign="top">0.71&#x2010;0.94</td><td align="left" valign="top">.004</td><td align="left" valign="top">1.31</td><td align="left" valign="top">1.13&#x2010;1.52</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Nonurban</td><td align="left" valign="top">1.02</td><td align="left" valign="top">0.89&#x2010;1.18</td><td align="left" valign="top">.73</td><td align="left" valign="top">0.89</td><td align="left" valign="top">0.77&#x2010;1.03</td><td align="left" valign="top">.13</td></tr><tr><td align="left" valign="top">Female &#x00D7; Black</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.43&#x2010;1.02</td><td align="left" valign="top">.06</td><td align="left" valign="top">NE</td><td align="left" valign="top">NE</td><td align="left" valign="top">.50</td></tr><tr><td align="left" valign="top">Female &#x00D7; Latine</td><td align="left" valign="top">52.99</td><td align="left" valign="top">23.29&#x2010;120.58</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">NE</td><td align="left" valign="top">NE</td><td align="left" valign="top">.62</td></tr><tr><td align="left" valign="top">Female &#x00D7; Asian</td><td align="left" valign="top">0.88</td><td align="left" valign="top">0.62&#x2010;1.25</td><td align="left" valign="top">.47</td><td align="left" valign="top">NE</td><td align="left" valign="top">NE</td><td align="left" valign="top">.59</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Reference categories: male (gender), White (race and ethnicity). MDD interaction terms exhibited quasi-complete separation due to multiple models producing zero diagnoses for specific gender &#x00D7; race combinations.</p></fn><fn id="table4fn2"><p><sup>b</sup>NE: not estimable due to quasi-complete separation.</p></fn><fn id="table4fn3"><p><sup>c</sup>SES: socioeconomic status.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>RQ2: Cross-Model Consistency and Divergence</title><p>For ambiguous vignettes, Fleiss &#x03BA; indicated moderate agreement (Fleiss &#x03BA;=0.410, 95% CI 0.397&#x2010;0.422). Agreement varied by demographic subgroup, with highest agreement for Latine patients (Fleiss &#x03BA;=0.477) and lowest for Asian patients (Fleiss &#x03BA;=0.127). Agreement was higher for female patients (Fleiss &#x03BA;=0.453) than male patients (Fleiss &#x03BA;=0.307; <xref ref-type="table" rid="table5">Table 5</xref>). Pairwise Cohen &#x03BA; ranged from 0.010 (Mistral Advanced vs Mistral Base) to 0.835 (DeepSeek Advanced vs ChatGPT Advanced), and models within the same family did not consistently show higher agreement than models across families (<xref ref-type="table" rid="table6">Table 6</xref>).</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Fleiss &#x03BA; agreement by demographic subgroup (ambiguous vignettes)<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variable and subgroup</td><td align="left" valign="bottom">Fleiss &#x03BA;</td><td align="left" valign="bottom">Vignettes<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup>, n</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Race and ethnicity</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>White</td><td align="left" valign="top">0.412</td><td align="left" valign="top">142</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Black</td><td align="left" valign="top">0.297</td><td align="left" valign="top">142</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Latine</td><td align="left" valign="top">0.477</td><td align="left" valign="top">144</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Asian</td><td align="left" valign="top">0.127</td><td align="left" valign="top">136</td></tr><tr><td align="left" valign="top" colspan="3">Sex</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">0.307</td><td align="left" valign="top">284</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">0.453</td><td align="left" valign="top">280</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>&#x03BA; values calculated across all 10 large language model configurations for each subgroup.</p></fn><fn id="table5fn2"><p><sup>b</sup>Unique vignettes in subgroup.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Pairwise Cohen &#x03BA; agreement matrix across 10 large language model configurations (ambiguous vignettes).</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model configuration</td><td align="left" valign="bottom">CGb<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup></td><td align="left" valign="bottom">CGa<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup></td><td align="left" valign="bottom">DSb<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup></td><td align="left" valign="bottom">DSa<sup><xref ref-type="table-fn" rid="table6fn4">d</xref></sup></td><td align="left" valign="bottom">LLb<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup></td><td align="left" valign="bottom">LLa<sup><xref ref-type="table-fn" rid="table6fn6">f</xref></sup></td><td align="left" valign="bottom">MIb<sup><xref ref-type="table-fn" rid="table6fn7">g</xref></sup></td><td align="left" valign="bottom">MIa<sup><xref ref-type="table-fn" rid="table6fn8">h</xref></sup></td><td align="left" valign="bottom">CLb<sup><xref ref-type="table-fn" rid="table6fn9">i</xref></sup></td><td align="left" valign="bottom">CLa<sup><xref ref-type="table-fn" rid="table6fn10">j</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">ChatGPT Base</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table6fn11">k</xref></sup></td><td align="left" valign="top">0.540</td><td align="left" valign="top">0.594</td><td align="left" valign="top">0.583</td><td align="left" valign="top">0.463</td><td align="left" valign="top">0.317</td><td align="left" valign="top">0.389</td><td align="left" valign="top">0.447</td><td align="left" valign="top">0.439</td><td align="left" valign="top">0.577</td></tr><tr><td align="left" valign="top">ChatGPT Adv</td><td align="left" valign="top">0.540</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.698</td><td align="left" valign="top">0.835</td><td align="left" valign="top">0.290</td><td align="left" valign="top">0.318</td><td align="left" valign="top">0.260</td><td align="left" valign="top">0.400</td><td align="left" valign="top">0.532</td><td align="left" valign="top">0.617</td></tr><tr><td align="left" valign="top">DeepSeek Base</td><td align="left" valign="top">0.594</td><td align="left" valign="top">0.698</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.695</td><td align="left" valign="top">0.388</td><td align="left" valign="top">0.313</td><td align="left" valign="top">0.352</td><td align="left" valign="top">0.363</td><td align="left" valign="top">0.612</td><td align="left" valign="top">0.576</td></tr><tr><td align="left" valign="top">DeepSeek Adv</td><td align="left" valign="top">0.583</td><td align="left" valign="top">0.835</td><td align="left" valign="top">0.695</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.294</td><td align="left" valign="top">0.289</td><td align="left" valign="top">0.266</td><td align="left" valign="top">0.417</td><td align="left" valign="top">0.531</td><td align="left" valign="top">0.626</td></tr><tr><td align="left" valign="top">Llama Base</td><td align="left" valign="top">0.463</td><td align="left" valign="top">0.290</td><td align="left" valign="top">0.388</td><td align="left" valign="top">0.294</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.196</td><td align="left" valign="top">0.393</td><td align="left" valign="top">0.315</td><td align="left" valign="top">0.326</td><td align="left" valign="top">0.335</td></tr><tr><td align="left" valign="top">Llama Adv</td><td align="left" valign="top">0.317</td><td align="left" valign="top">0.318</td><td align="left" valign="top">0.313</td><td align="left" valign="top">0.289</td><td align="left" valign="top">0.196</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.168</td><td align="left" valign="top">0.218</td><td align="left" valign="top">0.321</td><td align="left" valign="top">0.514</td></tr><tr><td align="left" valign="top">Mistral Base</td><td align="left" valign="top">0.389</td><td align="left" valign="top">0.260</td><td align="left" valign="top">0.352</td><td align="left" valign="top">0.266</td><td align="left" valign="top">0.393</td><td align="left" valign="top">0.168</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.010</td><td align="left" valign="top">0.100</td><td align="left" valign="top">0.286</td></tr><tr><td align="left" valign="top">Mistral Adv</td><td align="left" valign="top">0.447</td><td align="left" valign="top">0.400</td><td align="left" valign="top">0.363</td><td align="left" valign="top">0.417</td><td align="left" valign="top">0.315</td><td align="left" valign="top">0.218</td><td align="left" valign="top">0.010</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.477</td><td align="left" valign="top">0.484</td></tr><tr><td align="left" valign="top">Claude Base</td><td align="left" valign="top">0.439</td><td align="left" valign="top">0.532</td><td align="left" valign="top">0.612</td><td align="left" valign="top">0.531</td><td align="left" valign="top">0.326</td><td align="left" valign="top">0.321</td><td align="left" valign="top">0.100</td><td align="left" valign="top">0.477</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.473</td></tr><tr><td align="left" valign="top">Claude Adv</td><td align="left" valign="top">0.577</td><td align="left" valign="top">0.617</td><td align="left" valign="top">0.576</td><td align="left" valign="top">0.626</td><td align="left" valign="top">0.335</td><td align="left" valign="top">0.514</td><td align="left" valign="top">0.286</td><td align="left" valign="top">0.484</td><td align="left" valign="top">0.473</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>CGb: ChatGPT Base. </p></fn><fn id="table6fn2"><p><sup>b</sup>CGa: ChatGPT Advanced.</p></fn><fn id="table6fn3"><p><sup>c</sup>DSb: DeepSeek Base.</p></fn><fn id="table6fn4"><p><sup>d</sup>DSa: DeepSeek Advanced.</p></fn><fn id="table6fn5"><p><sup>e</sup>LLb: Llama Base.</p></fn><fn id="table6fn6"><p><sup>f</sup>LLa: Llama Advanced.</p></fn><fn id="table6fn7"><p><sup>g</sup>MIb: Mistral Base.</p></fn><fn id="table6fn8"><p><sup>h</sup>MIa: Mistral Advanced.</p></fn><fn id="table6fn9"><p><sup>i</sup>CLb: Claude Base.</p></fn><fn id="table6fn10"><p><sup>j</sup>CLa: Claude Advanced.</p></fn><fn id="table6fn11"><p><sup>k</sup>Self-comparison.</p></fn></table-wrap-foot></table-wrap><p>Cochran Q tests revealed significant heterogeneity in diagnosis rates across LLMs for both AN (Q<sub>9</sub>=388.44, <italic>P</italic>&#x003C;.001; range 12.9%&#x2010;45.6%) and MDD (Q<sub>9</sub>=764.59, <italic>P</italic>&#x003C;.001; range 24.6%&#x2010;62.2%). Post hoc pairwise McNemar tests identified 37 of 45 pairwise comparisons as significantly different after Benjamini-Hochberg correction. Notably, the largest within-family divergences occurred for Claude (AN rates: 45.6% base vs 12.9% advanced) and Llama (MDD rates: 62.2% base vs 24.6% advanced), suggesting that fine-tuning and alignment procedures substantially alter diagnostic behavior. All 45 pairwise McNemar post hoc comparisons are reported in Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p><xref ref-type="fig" rid="figure3">Figure 3</xref> illustrates the per-model ORs for race and ethnicity effects on AN diagnosis, revealing substantial heterogeneity across LLMs.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Race and ethnicity effects on anorexia nervosa diagnosis by large language model: per-model odds ratios with 95% CIs (ambiguous vignettes). The dashed line indicates odds ratio 1.0 (no effect).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e93498_fig03.png"/></fig><p>Pairwise Cohen &#x03BA; values ranged from 0.010 (Mistral Base vs Mistral Advanced) to 0.835 (ChatGPT Advanced vs DeepSeek Advanced; <xref ref-type="table" rid="table6">Table 6</xref>).</p></sec><sec id="s3-5"><title>Robustness Checks</title><p>Pooled logistic regression without random effects produced consistent patterns of significance, with odds ratios directionally aligned with the mixed-effects estimates, confirming that findings are robust to the modeling approach.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>Before interpreting these findings, it is worth delimiting the scope of the claims they support. The results establish that demographic bias arises in model outputs under the conditions tested here: across 10 configurations spanning 5 model families, when ambiguous eating-disorder vignettes were presented under a forced single-diagnosis format with clinical content held constant. The consistency of the racial effects across all 10 configurations supports generalization across contemporary models rather than implicating any single system. They do not, however, establish how these or other models would behave under different prompting strategies, in multiturn or open-ended clinical interactions, or for diagnostic scenarios beyond the AN presentations examined here. We therefore frame the following discussion as characterizing demographic bias under a specific, controlled evaluation paradigm, and distinguish this from broader claims about LLM behavior across the full range of clinical use, which the present design cannot adjudicate.</p><p>This study examined whether LLMs exhibit systematic diagnostic biases when evaluating clinical vignettes for eating disorders. Two key findings emerged. First, under the tested conditions, LLMs demonstrated demographic biases in psychiatric diagnosis, with Latine patients over 9 times more likely than White patients to receive an MDD diagnosis (OR 9.57, 95% CI 8.00&#x2013;11.45), Black patients over 6 times more likely (OR 6.09, 95% CI 5.13&#x2013;7.24), and Asian patients nearly 3 times more likely to receive an AN diagnosis (OR 2.88, 95% CI 2.44&#x2013;3.42) when clinical presentations were identical. These effects were consistent across the majority of individual models, with racial bias in MDD diagnosis present in all 10 LLM configurations tested.</p><p>Second, intermodel agreement was only moderate for ambiguous cases (Fleiss &#x03BA;=0.410, 95% CI 0.397&#x2010;0.422), indicating substantial variability in diagnostic behavior across LLM configurations. This variability was pronounced even within model families: Claude Base and Claude Advanced produced AN diagnosis rates of 45.6% and 12.9%, respectively, while Llama Base and Llama Advanced produced MDD rates of 62.2% and 24.6%, indicating that tier and alignment choices within a single family can substantially alter diagnostic behavior.</p><p>Within the experimental setting, these input-output effects can be attributed to demographic framing by virtue of the matched-pair design, though the internal mechanisms that generate them lie beyond what the design can resolve. Rather than indicating LLMs as unsuitable for clinical contexts, these findings reveal specific, measurable patterns that can inform targeted improvements to training procedures, model architectures, and deployment frameworks. In principle, LLMs could be developed not merely to replicate human decision-making but to improve upon it, although this study did not compare model and clinician performance and so cannot establish that current models approach, let alone exceed, human judgment. If human clinicians exhibit diagnostic biases, as decades of research confirm [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref21">21</xref>], the goal should be developing AI systems that actively mitigate against these biases rather than perpetuating them. By making bias patterns visible and quantifiable, this research provides a foundation for working toward more equitable systems. Realizing that potential, however, would require bias-mitigation strategies that do not yet exist; as tested, the models examined here reproduced rather than reduced demographic disparities.</p><p>All ambiguous vignettes were constructed to present clinical features primarily consistent with AN while permitting plausible differential diagnosis, mirroring the diagnostic uncertainty encountered in real-world clinical practice. Systematic demographic shifts toward MDD in these cases therefore are consistent with systematic influence of patient identity on diagnostic output rather than differences in clinical presentation.</p></sec><sec id="s4-2"><title>Mechanisms of Demographic Bias</title><p>A note on inference is warranted before considering mechanisms. Because clinical content was held constant within each matched pair and only the demographic header varied, the design supports inference of a causal effect of demographic cues on the observed diagnostic shifts within the experimental setting: the change in output can be ascribed to the change in patient identity rather than to differences in clinical presentation. It does not, however, identify the internal processes through which models produce these shifts. The mechanisms discussed in this section are therefore advanced as hypotheses consistent with the observed input-output relationships, not as pathways the present design can directly observe or confirm. The two most probable sources of the demographic biases observed are: (1) training data composition and (2) algorithmic architecture and optimization processes. Both represent modifiable factors rather than inherent limitations of the technology.</p><p>Concerning training data, LLMs learn statistical associations from text corpora that reflect historical patterns of diagnosis and treatment [<xref ref-type="bibr" rid="ref17">17</xref>]. If Black patients have been disproportionately diagnosed with depression in clinical literature, case studies, and electronic health records, models may learn these associations regardless of symptom presentation. An analogous mechanism may operate here, paralleling the finding by Obermeyer et al [<xref ref-type="bibr" rid="ref11">11</xref>] that health care algorithms trained on historical data encoded racial bias through proxy variables. The critical insight from that work, however, was that once identified, such biases could be substantially reduced through targeted data curation and algorithmic adjustment. The same principle applies here: if training data composition contributes to diagnostic bias, this suggests clear intervention points</p><p>These findings underscore the need for closer attention to training data in clinical AI development. Current LLM training pipelines prioritize scale, drawing from vast internet corpora with limited curation for clinical accuracy or demographic balance. The biases observed in this study are a plausible consequence of this approach: models trained predominantly on text reflecting historical diagnostic patterns may reproduce those patterns, even when they conflict with evidence-based clinical standards. Moving forward, developers of clinical AI systems should prioritize training data quality over quantity, implementing systematic audits for demographic representation, stereotype content, and alignment with current diagnostic criteria. The goal should not be models that mirror the clinical literature as it exists, but models that reflect clinical best practices as they should be applied.</p><p>The finding that Asian patients were significantly more likely to receive AN diagnoses is less readily explained by clinical stereotypes and illustrates how training data can encode both underrepresentation and misrepresentation. Research has documented eating pathology among Asian and Asian American populations [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]; however, the dominant narrative associates AN with White patients [<xref ref-type="bibr" rid="ref20">20</xref>], who comprise approximately 70% of eating disorder research samples [<xref ref-type="bibr" rid="ref32">32</xref>], which reinforces imbalanced learned associations. Alternatively, model outputs may reflect cultural stereotypes present in training text [<xref ref-type="bibr" rid="ref33">33</xref>]. LLMs trained on broad internet corpora may encode associations between Asian identity and thinness or dietary restriction that, while not originating from clinical literature, may influence diagnostic reasoning when demographic cues are present. Either mechanism suggests that careful auditing of training data for demographic balance and stereotype content could mitigate these patterns. The socioeconomic bias observed, in which low-SES patients were less likely to receive AN diagnoses (OR 0.61, 95% CI 0.53&#x2013;0.70) and more likely to receive MDD diagnoses (OR 1.41, 95% CI 1.24&#x2013;1.59), mirrors a well-documented clinical stereotype in which eating disorders are perceived as conditions affecting predominantly affluent individuals [<xref ref-type="bibr" rid="ref20">20</xref>]. This finding is notable because SES was conveyed through a single sentence in the vignette header, yet was sufficient to shift diagnostic output. If LLMs encode the assumption that low-SES patients do not develop eating disorders, they risk replicating the same access and detection barriers that already contribute to underdiagnosis in lower-income populations.</p><p>Algorithmic factors, including attention mechanisms, reinforcement learning from human feedback (RLHF), and optimization objectives, may also contribute to observed biases [<xref ref-type="bibr" rid="ref33">33</xref>]. Notably, RLHF relies on human annotators whose judgments may encode the same biases present in clinical practice. If annotators rate outputs as &#x201C;higher quality&#x201D; when they align with stereotypical expectations, the model may learn to reproduce those stereotypes. This observation points to an underexplored opportunity: RLHF could be deliberately designed to penalize demographically biased outputs rather than inadvertently rewarding them. Such an approach could position LLMs not as mirrors of human judgment but as tools calibrated to counteract known human biases.</p><p>The finding that female patients were less likely than male patients to receive both AN (OR 0.43, 95% CI 0.37&#x2013;0.49) and MDD (OR 0.50, 95% CI 0.44&#x2013;0.57) diagnoses warrants particular attention, as it partially contradicts patterns documented in the human clinician literature. Research suggests that clinicians overdiagnose depression in women [<xref ref-type="bibr" rid="ref21">21</xref>] and that the prototypical AN patient is female [<xref ref-type="bibr" rid="ref20">20</xref>], yet LLMs in this study favored neither diagnosis for female patients. Several factors may explain this divergence. First, the forced single-diagnosis paradigm creates a zero-sum dynamic: because each vignette produces exactly one diagnosis, racial and ethnic effects may dominate the diagnostic output, effectively suppressing gender effects that would emerge in a multidiagnosis or differential-diagnosis format. The gender &#x00D7; race or ethnicity interaction for AN supports this interpretation: the reduced AN diagnosis rate for females was consistent across White, Asian, and Black patients, but reversed entirely for Latine patients (interaction OR 52.99, 95% CI 23.29&#x2010;120.58; this estimate is imprecise owing to cell sparsity, so the directional reversal, rather than the exact magnitude, is the interpretable result), for whom males received virtually no AN diagnoses. This pattern suggests that LLMs applied compounding demographic heuristics in which Latine ethnicity and male gender may have jointly suppressed AN diagnosis. More broadly, because this forced-choice format compels a categorical diagnosis rather than permitting graded or differential reasoning, it may amplify the apparent magnitude of demographic differences relative to naturalistic settings; the direction of the observed effects is unlikely to be an artifact of the constraint, since it is held identical within every matched pair, but the absolute magnitudes should be interpreted as specific to the forced-choice paradigm and may not transfer directly to contexts that permit differential diagnosis.</p><p>Second, LLM training data may encode a corrective signal. Research on RLHF&#x2019;s effects on demographic representation suggests a possible mechanism: studies comparing pre- and postalignment models have found that RLHF can overcorrect for gender representation, producing outputs that overrepresent female perspectives while leaving underlying stereotypical associations intact [<xref ref-type="bibr" rid="ref34">34</xref>]. If similar overcorrection operates in clinical contexts, RLHF procedures designed to reduce gender stereotyping may suppress clinically valid gender-linked base rates, producing outputs that are demographically neutral at the cost of diagnostic sensitivity. This hypothesis warrants direct testing by comparing base and RLHF-aligned versions of the same model on clinical tasks. More broadly, awareness of gender bias in psychiatric diagnosis has generated substantial commentary in clinical and popular literature, and models trained on this text may have learned to counteract rather than reproduce gender stereotypes, even while remaining susceptible to racial biases that have received comparatively less corrective attention in training corpora. Third, the per-model results revealed that the direction of gender bias was not consistent across LLMs: Llama Base and Mistral Base diagnosed females with AN at higher rates than males, while most other models showed the opposite pattern. This inconsistency suggests that gender bias in LLMs is more labile than racial bias, which was directionally consistent across all 10 configurations.</p></sec><sec id="s4-3"><title>Interpreting Cross-Model Variation</title><p>The moderate intermodel agreement for ambiguous vignettes (Fleiss &#x03BA;=0.410) reflects differences in training corpora, optimization objectives, and alignment procedures across model families. Rather than viewing this variation as problematic, it offers valuable information about which approaches produce more or less biased outputs. AN diagnosis rates on ambiguous vignettes ranged from 12.9% (Claude Advanced) to 45.6% (Claude Base), while MDD rates ranged from 24.6% (Llama Advanced) to 62.2% (Llama Base), demonstrating that model behavior is highly sensitive to design choices, suggesting substantial room for optimization. This pattern echoes prior diagnostic-classification work: when ChatGPT, Claude, and Gemini were evaluated on identifying MDD from clinical vignettes, accuracy ranged from near chance to near-perfect across families [<xref ref-type="bibr" rid="ref6">6</xref>], indicating that wide cross-model divergence on vignette-based psychiatric classification is not unique to the present design.</p><p>The lowest agreement occurred for Asian patients (Fleiss &#x03BA;=0.127), revealing that current models lack stable, reliable representations for this demographic group in eating disorder contexts. This pattern likely reflects heterogeneity or sparsity in how Asian populations appear across training corpora. From an improvement perspective, this finding identifies a specific gap that data augmentation, targeted fine-tuning, or ensemble approaches might address. Notably, the variation itself is informative: some models performed better than others for this subgroup, and understanding what differentiates higher-performing models could guide development of more equitable systems.</p><p>Within-family divergence was particularly striking. Claude Base and Claude Advanced produced AN diagnosis rates of 45.6% and 12.9%, respectively, while Mistral Base and Mistral Advanced showed near-zero pairwise agreement (Cohen &#x03BA;=0.010). These patterns indicate that posttraining alignment and fine-tuning procedures do not merely refine base model behavior but can fundamentally alter diagnostic profiles. This finding has direct implications for clinical deployment: selecting a &#x201C;more advanced&#x201D; version of a model does not guarantee reduced bias, and may in some cases introduce new bias patterns. More broadly, these within-family divergences, like the intersectional effects reported above, would be invisible in analyses examining only main effects, underscoring the importance of disaggregated evaluation in clinical AI auditing.</p></sec><sec id="s4-4"><title>Limitations</title><p>Several limitations contextualize these findings. First, this study used text-based vignettes rather than real clinical encounters, which simplifies the complexity of psychiatric assessment. Actual diagnostic processes involve nonverbal cues, longitudinal patient history, and clinical intuition that vignettes cannot capture. While vignette-based methods are well-established in bias research [<xref ref-type="bibr" rid="ref26">26</xref>], findings should be validated in more naturalistic settings as LLMs are integrated into clinical workflows. Additionally, the forced single-diagnosis prompt format (&#x201C;respond with just the diagnosis name&#x201D;) does not reflect how clinicians typically use LLMs in practice, where open-ended queries, differential diagnoses, and chain-of-thought reasoning are more common. That said, forced single-diagnosis outputs are not uncommon in applied settings, where automated intake systems, triage chatbots, and clinical decision support tools frequently require categorical classification. This constraint may amplify observed biases by eliminating the hedging and differential reasoning that LLMs produce in more naturalistic contexts; the present design should therefore be read as a conservative test of whether demographic information influences diagnostic output rather than a simulation of clinical LLM use. Future research using differential-diagnosis prompts, in which models list multiple possibilities with associated confidence, could assess whether bias attenuates once the forced-choice constraint is removed.</p><p>Second, all vignettes were derived from a single target condition, AN, and although the ambiguous presentations admitted a broader differential (most often MDD, but also ARFID and other feeding and eating disorders), findings may not generalize to other psychiatric conditions or clinical contexts. The magnitude and direction of bias likely vary across conditions with different demographic distributions in training data, and comprehensive auditing should span multiple diagnostic categories.</p><p>Third, the data were collected across 2 periods, and the configurations were not uniformly version-stable. The Claude models were recollected in February 2026 to correct a temperature parameter error identified during manuscript preparation; both were accessed via version-pinned end points (claude-3-haiku-20240307 and claude-sonnet-4&#x2010;20250514), so the underlying weights were identical regardless of query date, although provider-side changes to system prompts or safety filters between October 2025 and February 2026 cannot be excluded. The consistency of bias patterns across the 8 models collected in October 2025 suggests the gap had minimal impact on overall findings. One model, Mistral Advanced, was accessed via a non&#x2013;version-pinned end point (mistral-large-latest), so its underlying version cannot be retrospectively confirmed.</p><p>Fourth, the models evaluated represent a snapshot of a rapidly changing field, and the set tested did not include every system available during the collection window. The GPT-5 family had been released before data collection began but was not incorporated, because the data-collection pipeline was implemented and run on the prior generation of GPT and was not updated to add GPT-5 before collection concluded; as the evaluation could not be rerun within the study timeline, GPT-5 is absent from the present analysis. Google&#x2019;s Gemini models were likewise not tested, owing to an access constraint that prevented reliable programmatic data collection during the study window rather than to a selection preference. The study&#x2019;s emphasis on widely used, lower-cost configurations (see &#x201C;Selection Rationale&#x201D; section) governs which models were actively sought, but it does not account for the GPT-5 omission, which represents a limitation of the data-collection process. Future evaluations should incorporate more recent frontier systems to determine whether the observed biases persist. Model behavior also evolves as providers retrain and update systems: version-pinned end points can be retired (as has since occurred for the Claude configurations used here), unpinned end points can change without notice, and provider-side updates can alter outputs even when weights are nominally fixed. Results should therefore be read as characterizing these models at the time of testing, and replication on current systems is needed to determine whether the observed biases persist, attenuate, or shift. This transience also motivates the repeated, version-aware auditing the present methodology supports, offering a way to track whether demographic bias decreases as models are updated.</p><p>Fifth, the moderate intermodel agreement for ambiguous vignettes (Fleiss &#x03BA;=0.410, 95% CI 0.397&#x2010;0.422) warrants careful interpretation. This statistic indexes agreement among the 10 configurations, not the diagnostic competence of any individual model, so low agreement is not itself evidence of poor performance. Agreement on the control vignettes was near-unanimous, indicating that the models classified unambiguous cases consistently; divergence was confined to the ambiguous condition, which was deliberately designed to permit several defensible differential diagnoses. Under such conditions, disagreement across models is expected rather than anomalous&#x2014;much as clinicians frequently disagree on ambiguous cases&#x2014;and the cross-model variability is better understood as a substantive finding (RQ2) than a deficiency. We nonetheless acknowledge that this variability means model choice has direct consequences for diagnostic output, reinforcing the need for model-specific validation before clinical deployment.</p><p>Finally, the 2 model families that assisted in formulating candidate vignette phrasings (ChatGPT and Claude) were also among those evaluated; for GPT-4o, the identical model served both roles, whereas the Claude version used during formulation (Claude Sonnet 4.5) differed from the tested Claude configurations (Claude 3 Haiku and Claude Sonnet 4). Because the final vignettes were validated by the research team and assembled by a deterministic script rather than generated by a model at test time, no configuration was evaluated on text it authored; nonetheless, subtle stylistic familiarity effects cannot be fully excluded, and future replications could formulate stimuli independently of any tested model. The study also did not examine the reasoning processes underlying diagnostic decisions, which future work using chain-of-thought or rationale-elicitation prompts could address.</p></sec><sec id="s4-5"><title>Implications</title><sec id="s4-5-1"><title>Research</title><p>These findings establish a methodology and baseline for ongoing bias monitoring in clinical LLMs. In doing so, they respond to a growing call for more rigorous and structured evaluation of clinically sensitive LLM systems: recent commentary has argued that psychiatric and other high-stakes applications require evaluation that extends beyond aggregate performance metrics to interrogate safety and reliability directly [<xref ref-type="bibr" rid="ref35">35</xref>], and a recent systematic review of LLM-based mental health tools found that the existing reporting of model configurations, calling for clinically grounded evaluation frameworks and transparent reporting of model and prompt settings [<xref ref-type="bibr" rid="ref36">36</xref>]. This study advances this agenda by pairing a controlled, counterfactual design with fully transparent reporting of model versions, prompts, and parameters (see &#x201C;LLM Selection and Configuration&#x201D; section; <xref ref-type="table" rid="table1">Table 1</xref>), providing an auditing approach that is both reproducible and demographically disaggregated. Researchers should extend this approach to other psychiatric conditions, particularly those with known diagnostic disparities (eg, attention-deficit/hyperactivity disorder [ADHD], bipolar disorder, and schizophrenia), building a comprehensive map of where biases emerge and how they vary across conditions. Longitudinal studies tracking model behavior across version updates could determine whether bias is decreasing as developers implement fairness interventions, creating accountability mechanisms that incentivize improvement.</p><p>A critical question for future research is whether LLMs merely reflect human biases present in training data or actively amplify them. Studies comparing LLM bias magnitude to documented human clinician bias on matched tasks could disentangle these mechanisms and inform intervention strategies.</p><p>The low agreement for Asian patients highlights the importance of disaggregated analysis in AI fairness research. Aggregate performance metrics can mask substantial subgroup heterogeneity, and researchers should routinely report demographic stratification. Additionally, the finding that capability tier did not predict bias patterns suggests that evaluation frameworks should assess fairness independently from accuracy or reasoning benchmarks. Collaboration between clinical researchers and AI developers could accelerate progress: the patterns identified here provide specific targets for intervention, and partnerships that combine clinical expertise with technical capacity for model modification offer the most promising path toward equitable systems.</p></sec><sec id="s4-5-2"><title>Clinical Practice</title><p>Translating these findings into practice requires shifting focus from what the models do to the human and institutional context in which they would be used. For clinicians and health care systems, these findings support thoughtful integration rather than wholesale adoption or rejection of LLM tools. It bears emphasis, however, that the models evaluated here are not currently suitable for unsupervised diagnostic use. The biases observed were large, directionally consistent across all 10 configurations for racial effects, and sufficient to misroute care for minority and male patients on the basis of identity alone; deploying such models for diagnosis without robust bias auditing and human oversight would risk reproducing, and potentially amplifying at scale, the very disparities documented here. These consequences are ultimately borne by patients rather than by systems: a diagnostic shift away from AN for a Black, Latine, male, or lower-income patient is not an abstract error but a missed or delayed opportunity for appropriate eating-disorder care, falling hardest on the very groups already most likely to be overlooked under existing clinical practice. Keeping the affected patient and the clinician accountable for the decision at the center of evaluation is therefore essential as these tools move toward practice.</p><p>Even so, the biases observed were not uniform across models, suggesting that careful model selection and validation could identify configurations suitable for specific clinical contexts; the constructive possibilities below are therefore conditional on mitigation and monitoring that are not yet standard practice. The variation in diagnosis rates across models (12.9% to 45.6% for AN and 24.6% to 62.2% for MDD on ambiguous vignettes) underscores that model selection has direct implications for patient care. Health care organizations implementing LLM-based tools should establish bias auditing protocols, with particular attention to demographic subgroups underrepresented in model training. Such auditing is best understood not as a single institution&#x2019;s responsibility but as a shared governance obligation distributed across model developers, deploying health care organizations, and regulators, each of whom holds a different lever over how these systems are built, selected, and monitored. Transparent, standardized reporting of model versions, prompts, and evaluation results&#x2014;of the kind this study provides and that recent reviews of clinical LLM tools have called for [<xref ref-type="bibr" rid="ref36">36</xref>]&#x2014;is a precondition for such accountability, allowing bias to be detected and attributed before deployment rather than after harm has occurred. Transparency about which models are deployed and how they were evaluated should thus become standard practice, enabling clinicians to contextualize LLM outputs appropriately, and to calibrate trust based on known model performance for specific patient populations rather than treating all LLM suggestions with uniform skepticism. This caution aligns with broader work on mental-health AI: in a comparative study of traditional machine learning classifiers and a zero-shot LLM on social-media mental health classification, the language model&#x2019;s free-text rationales were more readable than feature-based explanations, yet less faithful to the underlying decision, such that fluent justifications could mask brittle reasoning [<xref ref-type="bibr" rid="ref37">37</xref>]. Although that work addressed a different task and did not examine demographic bias, its conclusion converges with ours: mental-health LLM outputs are best treated as screening or triage signals subject to human oversight rather than as autonomous diagnostic judgments.</p><p>Ultimately, within the paradigm tested, these findings suggest a longer-term possibility that LLMs could be reoriented from bias-propagation tools to bias-detection tools, alerting clinicians when demographic factors may be inappropriately influencing diagnostic reasoning. Achieving this vision requires intentional design: LLMs must be trained not merely on clinical text as it exists but on clinical reasoning as it should be applied.</p></sec></sec></sec></body><back><ack><p>Generative AI tools were used in both the research methods and the preparation of this manuscript. GPT-4o (OpenAI) and Claude Sonnet 4.5 (Anthropic) were used to draft and refine control and ambiguous vignette narratives from author-specified content; all vignette texts were reviewed, edited, and validated by the research team against <italic>DSM-5</italic> (<italic>Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition</italic>) criteria, and the final corpus was assembled by a deterministic script rather than model-generated. Claude (Anthropic) was additionally used to support drafting and editing of the manuscript, including the abstract, and to format and summarize statistical outputs. All AI-assisted content was reviewed, verified, and edited by the authors, who take full responsibility for the accuracy, originality, and integrity of all content. Generative AI was not used to generate data, conduct analyses, or produce results, and is not listed as an author.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed during this study, along with the analysis code, are available in the Open Science Framework repository [<xref ref-type="bibr" rid="ref38">38</xref>]</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: DM, LG</p><p>Data curation: DM, BL, SJ, JJ, CS</p><p>Formal analysis: DM, CS, CD</p><p>Investigation: DM, BL, SJ, JJ</p><p>Methodology: DM, BL, SJ, JJ</p><p>Project administration: DM</p><p>Supervision: CD</p><p>Writing &#x2013; original draft: DM</p><p>Writing &#x2013; review &#x0026; editing: DM, BL, SJ, JJ, LG, CD</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ADHD</term><def><p>attention-deficit/hyperactivity disorder</p></def></def-item><def-item><term id="abb2">AN</term><def><p>anorexia nervosa</p></def></def-item><def-item><term id="abb3">ARFID</term><def><p>Avoidant/Restrictive Food Intake Disorder</p></def></def-item><def-item><term id="abb4"><italic>DSM-5</italic></term><def><p><italic>Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition</italic></p></def></def-item><def-item><term id="abb5">GenAI</term><def><p>generative AI</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">MDD</term><def><p>major depressive disorder</p></def></def-item><def-item><term id="abb8">OR</term><def><p>odds ratio</p></def></def-item><def-item><term id="abb9">RLHF</term><def><p>reinforcement learning from human feedback</p></def></def-item><def-item><term id="abb10">RQ</term><def><p>research question</p></def></def-item><def-item><term id="abb11">SES</term><def><p>socioeconomic status</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSJ</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Newby</surname><given-names>D</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>N</given-names> </name><name name-style="western"><surname>Joyce</surname><given-names>DW</given-names> </name><name name-style="western"><surname>Winchester</surname><given-names>LM</given-names> </name></person-group><article-title>Optimising the use of electronic medical records for large scale research in psychiatry</article-title><source>Transl Psychiatry</source><year>2024</year><month>06</month><day>1</day><volume>14</volume><issue>1</issue><fpage>232</fpage><pub-id pub-id-type="doi">10.1038/s41398-024-02911-1</pub-id><pub-id pub-id-type="medline">38824136</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lawrence</surname><given-names>HR</given-names> </name><name name-style="western"><surname>Schneider</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Rubin</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Matari&#x0107;</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>McDuff</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Jones Bell</surname><given-names>M</given-names> </name></person-group><article-title>The opportunities and risks of large language models in mental health</article-title><source>JMIR Ment Health</source><year>2024</year><month>07</month><day>29</day><volume>11</volume><fpage>e59479</fpage><pub-id pub-id-type="doi">10.2196/59479</pub-id><pub-id pub-id-type="medline">39105570</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fitzpatrick</surname><given-names>KK</given-names> </name><name name-style="western"><surname>Darcy</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vierhile</surname><given-names>M</given-names> </name></person-group><article-title>Delivering cognitive behavior therapy to young adults with symptoms of depression and anxiety using a fully automated conversational agent (Woebot): a randomized controlled trial</article-title><source>JMIR Ment Health</source><year>2017</year><month>06</month><day>6</day><volume>4</volume><issue>2</issue><fpage>e19</fpage><pub-id pub-id-type="doi">10.2196/mental.7785</pub-id><pub-id pub-id-type="medline">28588005</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Inkster</surname><given-names>B</given-names> </name><name name-style="western"><surname>Sarda</surname><given-names>S</given-names> </name><name name-style="western"><surname>Subramanian</surname><given-names>V</given-names> </name></person-group><article-title>An empathy-driven, conversational artificial intelligence agent (Wysa) for digital mental well-being: real-world data evaluation mixed-methods study</article-title><source>JMIR Mhealth Uhealth</source><year>2018</year><month>11</month><day>23</day><volume>6</volume><issue>11</issue><fpage>e12106</fpage><pub-id pub-id-type="doi">10.2196/12106</pub-id><pub-id pub-id-type="medline">30470676</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rensi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ni</surname><given-names>CF</given-names> </name><name name-style="western"><surname>Gonzalez, Jr.</surname><given-names>L</given-names> </name><name name-style="western"><surname>Barta</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dykeman</surname><given-names>C</given-names> </name><name name-style="western"><surname>Geisler</surname><given-names>J</given-names> </name></person-group><article-title>Evaluating generative AI for depression diagnosis: implications for counselor education and supervision</article-title><source>J Technol Couns Educ Superv</source><year>2025</year><volume>6</volume><issue>1</issue><fpage>Article</fpage><pub-id pub-id-type="doi">10.61888/2692-4129.1130</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zack</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lehman</surname><given-names>E</given-names> </name><name name-style="western"><surname>Suzgun</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Assessing the potential of GPT-4 to perpetuate racial and gender biases in health care: a model evaluation study</article-title><source>Lancet Digit Health</source><year>2024</year><month>01</month><volume>6</volume><issue>1</issue><fpage>e12</fpage><lpage>e22</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00225-X</pub-id><pub-id pub-id-type="medline">38123252</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bouguettaya</surname><given-names>A</given-names> </name><name name-style="western"><surname>Stuart</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Aboujaoude</surname><given-names>E</given-names> </name></person-group><article-title>Racial bias in AI-mediated psychiatric diagnosis and treatment: a qualitative comparison of four large language models</article-title><source>NPJ Digit Med</source><year>2025</year><month>06</month><day>4</day><volume>8</volume><issue>1</issue><fpage>332</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01746-4</pub-id><pub-id pub-id-type="medline">40467886</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Friedman</surname><given-names>B</given-names> </name><name name-style="western"><surname>Nissenbaum</surname><given-names>H</given-names> </name></person-group><article-title>Bias in computer systems</article-title><source>ACM Trans Inf Syst</source><year>1996</year><month>07</month><volume>14</volume><issue>3</issue><fpage>330</fpage><lpage>347</lpage><pub-id pub-id-type="doi">10.1145/230538.230561</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mehrabi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Morstatter</surname><given-names>F</given-names> </name><name name-style="western"><surname>Saxena</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lerman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Galstyan</surname><given-names>A</given-names> </name></person-group><article-title>A survey on bias and fairness in machine learning</article-title><source>ACM Comput Surv</source><year>2022</year><month>07</month><day>31</day><volume>54</volume><issue>6</issue><fpage>1</fpage><lpage>35</lpage><pub-id pub-id-type="doi">10.1145/3457607</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Obermeyer</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Powers</surname><given-names>B</given-names> </name><name name-style="western"><surname>Vogeli</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mullainathan</surname><given-names>S</given-names> </name></person-group><article-title>Dissecting racial bias in an algorithm used to manage the health of populations</article-title><source>Science</source><year>2019</year><month>10</month><day>25</day><volume>366</volume><issue>6464</issue><fpage>447</fpage><lpage>453</lpage><pub-id pub-id-type="doi">10.1126/science.aax2342</pub-id><pub-id pub-id-type="medline">31649194</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bolukbasi</surname><given-names>T</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>KW</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>JY</given-names> </name><name name-style="western"><surname>Saligrama</surname><given-names>V</given-names> </name><name name-style="western"><surname>Kalai</surname><given-names>AT</given-names> </name></person-group><article-title>Man is to computer programmer as woman is to homemaker? debiasing word embeddings</article-title><access-date>2026-08-31</access-date><conf-name>30th International Conference on Neural Information Processing Systems (NIPS 2016)</conf-name><conf-date>Dec 5-10, 2016</conf-date><fpage>4349</fpage><lpage>4357</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2016/hash/a486cd07e4ac3d270571622f4f316ec5-Abstract.html">https://proceedings.neurips.cc/paper/2016/hash/a486cd07e4ac3d270571622f4f316ec5-Abstract.html</ext-link></comment></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Caliskan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bryson</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>A</given-names> </name></person-group><article-title>Semantics derived automatically from language corpora contain human-like biases</article-title><source>Science</source><year>2017</year><month>04</month><day>14</day><volume>356</volume><issue>6334</issue><fpage>183</fpage><lpage>186</lpage><pub-id pub-id-type="doi">10.1126/science.aal4230</pub-id><pub-id pub-id-type="medline">28408601</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dixon</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sorensen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Thain</surname><given-names>N</given-names> </name><name name-style="western"><surname>Vasserman</surname><given-names>L</given-names> </name></person-group><article-title>Measuring and mitigating unintended bias in text classification</article-title><access-date>2026-08-21</access-date><conf-name>AIES &#x2019;18: Proceedings of the 2018 AAAI/ACM Conference on AI, Ethics, and Society</conf-name><conf-date>Feb 2-3, 2018</conf-date><conf-loc>New Orleans, LA, USA</conf-loc><fpage>67</fpage><lpage>73</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3278721">https://dl.acm.org/doi/proceedings/10.1145/3278721</ext-link></comment><pub-id pub-id-type="doi">10.1145/3278721.3278729</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pleiss</surname><given-names>G</given-names> </name><name name-style="western"><surname>Raghavan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Kleinberg</surname><given-names>J</given-names> </name><name name-style="western"><surname>Weinberger</surname><given-names>KQ</given-names> </name></person-group><article-title>On fairness and calibration</article-title><access-date>2026-08-31</access-date><conf-name>Advances in Neural Information Processing Systems 30 (NIPS 2017)</conf-name><conf-date>Dec 4-9, 2017</conf-date><conf-loc>Long Beach, CA, USA</conf-loc><fpage>5680</fpage><lpage>5689</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2017/hash/b8b9c74ac526fffbeb2d39ab038d1cd7-Abstract.html?">https://proceedings.neurips.cc/paper/2017/hash/b8b9c74ac526fffbeb2d39ab038d1cd7-Abstract.html?</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cary</surname><given-names>MP</given-names>  <suffix>Jr</suffix></name><name name-style="western"><surname>Zink</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Mitigating racial and ethnic bias and advancing health equity in clinical algorithms: a scoping review</article-title><source>Health Aff (Millwood)</source><year>2023</year><month>10</month><volume>42</volume><issue>10</issue><fpage>1359</fpage><lpage>1368</lpage><pub-id pub-id-type="doi">10.1377/hlthaff.2023.00553</pub-id><pub-id pub-id-type="medline">37782868</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bender</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Gebru</surname><given-names>T</given-names> </name><name name-style="western"><surname>McMillan-Major</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shmitchell</surname><given-names>S</given-names> </name></person-group><article-title>On the dangers of stochastic parrots: can language models be too big</article-title><conf-name>ACM Conference on Fairness, Accountability, and Transparency (FAccT '21)</conf-name><conf-date>Mar 3-10, 2021</conf-date><conf-loc>Virtual Event, Toronto, Canada</conf-loc><fpage>610</fpage><lpage>623</lpage><pub-id pub-id-type="doi">10.1145/3442188.3445922</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Insel</surname><given-names>TR</given-names> </name></person-group><article-title>The NIMH Research Domain Criteria (RDoC) Project: precision medicine for psychiatry</article-title><source>Am J Psychiatry</source><year>2014</year><month>04</month><volume>171</volume><issue>4</issue><fpage>395</fpage><lpage>397</lpage><pub-id pub-id-type="doi">10.1176/appi.ajp.2014.14020138</pub-id><pub-id pub-id-type="medline">24687194</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gara</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Minsky</surname><given-names>S</given-names> </name><name name-style="western"><surname>Silverstein</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Miskimen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Strakowski</surname><given-names>SM</given-names> </name></person-group><article-title>A naturalistic study of racial disparities in diagnoses at an outpatient behavioral health clinic</article-title><source>Psychiatr Serv</source><year>2019</year><month>02</month><day>1</day><volume>70</volume><issue>2</issue><fpage>130</fpage><lpage>134</lpage><pub-id pub-id-type="doi">10.1176/appi.ps.201800223</pub-id><pub-id pub-id-type="medline">30526340</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sonneville</surname><given-names>KR</given-names> </name><name name-style="western"><surname>Lipson</surname><given-names>SK</given-names> </name></person-group><article-title>Disparities in eating disorder diagnosis and treatment according to weight status, race/ethnicity, socioeconomic background, and sex among college students</article-title><source>Int J Eat Disord</source><year>2018</year><month>06</month><volume>51</volume><issue>6</issue><fpage>518</fpage><lpage>526</lpage><pub-id pub-id-type="doi">10.1002/eat.22846</pub-id><pub-id pub-id-type="medline">29500865</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bacigalupe</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mart&#x00ED;n</surname><given-names>U</given-names> </name><name name-style="western"><surname>Triolo</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Is the diagnosis and treatment of depression gender-biased? Evidence from a population-based aging cohort in Sweden</article-title><source>Int J Equity Health</source><year>2024</year><month>11</month><day>27</day><volume>23</volume><issue>1</issue><fpage>252</fpage><pub-id pub-id-type="doi">10.1186/s12939-024-02320-2</pub-id><pub-id pub-id-type="medline">39605074</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cavanagh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wilson</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Kavanagh</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Caputi</surname><given-names>P</given-names> </name></person-group><article-title>Differences in the expression of symptoms in men versus women with depression: a systematic review and meta-analysis</article-title><source>Harv Rev Psychiatry</source><year>2017</year><volume>25</volume><issue>1</issue><fpage>29</fpage><lpage>38</lpage><pub-id pub-id-type="doi">10.1097/HRP.0000000000000128</pub-id><pub-id pub-id-type="medline">28059934</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Carbonell</surname><given-names>&#x00C1;</given-names> </name><name name-style="western"><surname>Navarro-P&#x00E9;rez</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Mestre</surname><given-names>MV</given-names> </name></person-group><article-title>Challenges and barriers in mental healthcare systems and their impact on the family: a systematic integrative review</article-title><source>Health Soc Care Community</source><year>2020</year><month>09</month><volume>28</volume><issue>5</issue><fpage>1366</fpage><lpage>1379</lpage><pub-id pub-id-type="doi">10.1111/hsc.12968</pub-id><pub-id pub-id-type="medline">32115797</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Snowden</surname><given-names>LR</given-names> </name></person-group><article-title>Bias in mental health assessment and intervention: theory and evidence</article-title><source>Am J Public Health</source><year>2003</year><month>02</month><volume>93</volume><issue>2</issue><fpage>239</fpage><lpage>243</lpage><pub-id pub-id-type="doi">10.2105/ajph.93.2.239</pub-id><pub-id pub-id-type="medline">12554576</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Denecke</surname><given-names>K</given-names> </name><name name-style="western"><surname>Rivera-Romero</surname><given-names>O</given-names> </name><name name-style="western"><surname>L&#x00F3;pez-Campos</surname><given-names>G</given-names> </name><name name-style="western"><surname>Dorronzoro</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gabarron</surname><given-names>E</given-names> </name></person-group><article-title>Uncovering AI&#x2019;s hidden risks: an empirical analysis of health-related AI incidents and their ethical implications</article-title><source>AI Ethics</source><year>2026</year><month>04</month><volume>6</volume><issue>2</issue><fpage>169</fpage><pub-id pub-id-type="doi">10.1007/s43681-026-01012-7</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aguinis</surname><given-names>H</given-names> </name><name name-style="western"><surname>Bradley</surname><given-names>KJ</given-names> </name></person-group><article-title>Best practice recommendations for designing and implementing experimental vignette methodology studies</article-title><source>Organ Res Methods</source><year>2014</year><month>10</month><volume>17</volume><issue>4</issue><fpage>351</fpage><lpage>371</lpage><pub-id pub-id-type="doi">10.1177/1094428114547952</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kusner</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Silva</surname><given-names>R</given-names> </name><name name-style="western"><surname>Russell</surname><given-names>C</given-names> </name><name name-style="western"><surname>Loftus</surname><given-names>JR</given-names> </name></person-group><article-title>Counterfactual fairness</article-title><access-date>2026-08-26</access-date><conf-name>Advances in Neural Information Processing Systems 30 (NIPS 2017)</conf-name><conf-date>Dec 4-9, 2017</conf-date><conf-loc>Long Beach, CA, USA</conf-loc><fpage>4066</fpage><lpage>4077</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2017/hash/a486cd07e4ac3d270571622f4f316ec5-Abstract.html">https://proceedings.neurips.cc/paper/2017/hash/a486cd07e4ac3d270571622f4f316ec5-Abstract.html</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Russell</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kusner</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Loftus</surname><given-names>J</given-names> </name><name name-style="western"><surname>Silva</surname><given-names>R</given-names> </name></person-group><article-title>When worlds collide: integrating different counterfactual assumptions in fairness</article-title><access-date>2026-08-26</access-date><conf-name>Advances in Neural Information Processing Systems (NIPS 2017)</conf-name><conf-date>Dec 4-9, 2017</conf-date><conf-loc>Long Beach, CA, USA</conf-loc><fpage>6414</fpage><lpage>6423</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2017/hash/1271a7029c9df08643b631b02cf9e116-Abstract.html">https://proceedings.neurips.cc/paper/2017/hash/1271a7029c9df08643b631b02cf9e116-Abstract.html</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Kincaid</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Fishburne</surname><given-names>RP</given-names>  <suffix>Jr</suffix></name><name name-style="western"><surname>Rogers</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Chissom</surname><given-names>BS</given-names> </name></person-group><article-title>Derivation of new readability formulas (Automated Readability Index, Fog Count and Flesch Reading Ease Formula) for Navy enlisted personnel</article-title><year>1975</year><access-date>2026-08-26</access-date><publisher-name>Naval Technical Training Command, Research Branch</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://archive.org/details/DTIC_ADA006655">https://archive.org/details/DTIC_ADA006655</ext-link></comment><pub-id pub-id-type="doi">10.21236/ADA006655</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Uri</surname><given-names>RC</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>YK</given-names> </name><name name-style="western"><surname>Baker</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Munn-Chernoff</surname><given-names>MA</given-names> </name></person-group><article-title>Eating disorder symptoms in Asian American college students</article-title><source>Eat Behav</source><year>2021</year><month>01</month><volume>40</volume><fpage>101458</fpage><pub-id pub-id-type="doi">10.1016/j.eatbeh.2020.101458</pub-id><pub-id pub-id-type="medline">33307468</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Eating disorder risk and diagnosis among East Asian youth in the United States: findings from the Healthy Minds Study, 2020-2023</article-title><source>Int J Eat Disord</source><year>2026</year><month>03</month><volume>59</volume><issue>3</issue><fpage>595</fpage><lpage>601</lpage><pub-id pub-id-type="doi">10.1002/eat.24594</pub-id><pub-id pub-id-type="medline">41250962</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Egbert</surname><given-names>AH</given-names> </name><name name-style="western"><surname>Hunt</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Williams</surname><given-names>KL</given-names> </name><name name-style="western"><surname>Burke</surname><given-names>NL</given-names> </name><name name-style="western"><surname>Mathis</surname><given-names>KJ</given-names> </name></person-group><article-title>Reporting racial and ethnic diversity in eating disorder research over the past 20&#x2009;years</article-title><source>Int J Eat Disord</source><year>2022</year><month>04</month><volume>55</volume><issue>4</issue><fpage>455</fpage><lpage>462</lpage><pub-id pub-id-type="doi">10.1002/eat.23666</pub-id><pub-id pub-id-type="medline">34997609</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gallegos</surname><given-names>IO</given-names> </name><name name-style="western"><surname>Rossi</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Barrow</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Bias and fairness in large language models: a survey</article-title><source>Comput Linguist</source><year>2024</year><month>09</month><day>1</day><volume>50</volume><issue>3</issue><fpage>1097</fpage><lpage>1179</lpage><pub-id pub-id-type="doi">10.1162/coli_a_00524</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>E</given-names> </name><name name-style="western"><surname>Zhan</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>YB</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>HH</given-names> </name></person-group><article-title>More women, same stereotypes: unpacking the gender bias paradox in large language models</article-title><conf-name>34th ACM International Conference on Information and Knowledge Management (CIKM &#x2019;25)</conf-name><conf-date>Nov 10-14, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3746252">https://dl.acm.org/doi/proceedings/10.1145/3746252</ext-link></comment><pub-id pub-id-type="doi">10.1145/3746252.3760969</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name></person-group><article-title>Toward retrieval-grounded evaluation for conversational large language model-based risk assessment</article-title><source>JMIR AI</source><year>2026</year><month>03</month><day>12</day><volume>5</volume><fpage>e90759</fpage><pub-id pub-id-type="doi">10.2196/90759</pub-id><pub-id pub-id-type="medline">41818631</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cho</surname><given-names>HN</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>K</given-names> </name></person-group><article-title>Large language model-based chatbots and agentic AI for mental health counseling: systematic review of methodologies, evaluation frameworks, and ethical safeguards</article-title><source>JMIR AI</source><year>2026</year><month>03</month><day>13</day><volume>5</volume><fpage>e80348</fpage><pub-id pub-id-type="doi">10.2196/80348</pub-id><pub-id pub-id-type="medline">41592221</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Xie</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>Z</given-names> </name></person-group><article-title>Explainable AI for mental health detection from social media: a comparative study of traditional machine learning and a large language model</article-title><source>SSRN</source><comment>Preprint posted online on  Mar 19, 2026</comment><pub-id pub-id-type="doi">10.2139/ssrn.6429778</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>McCalla</surname><given-names>D</given-names> </name><name name-style="western"><surname>Li</surname><given-names>B</given-names> </name><name name-style="western"><surname>Jaeger</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Implicit bias in large language model diagnosis of eating disorders: experimental vignette study</article-title><source>JMIR Preprints</source><comment>Preprint posted online on  Feb 13, 2026</comment><pub-id pub-id-type="doi">10.2196/preprints.93498</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Per-model logistic regression odds ratios for anorexia nervosa and major depressive disorder diagnosis, Cochran rule compliance for chi-square tests, and pairwise McNemar post hoc comparisons across large language model configurations.</p><media xlink:href="ai_v5i1e93498_app1.docx" xlink:title="DOCX File, 23 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Per-model diagnostic output distributions by demographic subgroup for ambiguous vignettes across all 10 large language model configurations.</p><media xlink:href="ai_v5i1e93498_app2.pdf" xlink:title="PDF File, 1701 KB"/></supplementary-material></app-group></back></article>