<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e87730</article-id><article-id pub-id-type="doi">10.2196/87730</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Generative Large Language Models in Mental Health Care Settings: Systematic Review and Meta-Analysis</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Leung</surname><given-names>Janni</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Johnson</surname><given-names>Benjamin</given-names></name><degrees>BPsych (Hons)</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>McRae</surname><given-names>Kelsey</given-names></name><degrees>BPsych (Hons)</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Fong</surname><given-names>Stephanie</given-names></name><degrees>BSc (Hons)</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gonzalez</surname><given-names>Paula Cardona</given-names></name><degrees>BPsych (Hons)</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>McClure-Thomas</surname><given-names>Caitlin</given-names></name><degrees>BPsych (Hons)</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sun</surname><given-names>Tianze</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kern</surname><given-names>Naomi</given-names></name><degrees>BPsych (Hons)</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wong</surname><given-names>Yuen Ming</given-names></name><degrees>BPsych (Hons)</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Richard</given-names></name><degrees>MSW</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chan</surname><given-names>Gary Chung Kai</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>National Centre for Youth Substance Use Research, The University of Queensland</institution><addr-line>31 Upland Road</addr-line><addr-line>Brisbane</addr-line><addr-line>Queensland</addr-line><country>Australia</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Wang</surname><given-names>Liying</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Haun</surname><given-names>Markus W</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Benjamin Johnson, BPsych (Hons), National Centre for Youth Substance Use Research, The University of Queensland, 31 Upland Road, Brisbane, Queensland, 4067, Australia, 61 429894154; <email>ben.johnson@uq.net.au</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>31</day><month>8</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e87730</elocation-id><history><date date-type="received"><day>13</day><month>11</month><year>2025</year></date><date date-type="rev-recd"><day>23</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>24</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Janni Leung, Benjamin Johnson, Kelsey McRae, Stephanie Fong, Paula Cardona Gonzalez, Caitlin McClure-Thomas, Tianze Sun, Naomi Kern, Yuen Ming Wong, Richard Liu, Gary Chung Kai Chan. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 31.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e87730"/><abstract><sec><title>Background</title><p>General-purpose large language models (LLMs) are increasingly being tested in mental health care, where language is central to assessment, diagnosis, risk evaluation, therapeutic interaction, monitoring, and patient education. However, their clinical usefulness, safety, and readiness for implementation remain uncertain. Existing reviews have largely been descriptive or scoping in nature, and broad health care reviews have not examined in detail the distinctive risks and applications of LLMs in mental health care.</p></sec><sec><title>Objective</title><p>We aim to systematically review empirical evidence on the use of general-purpose LLMs in mental health care; characterize the clinical tasks, study designs, models, outcomes, and methodological quality of the evidence; and synthesize findings across clinically meaningful task domains, including quantitative synthesis where sufficiently comparable studies were available.</p></sec><sec sec-type="methods"><title>Methods</title><p>We followed PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) and searched PubMed, Embase, ACM Digital Library, IEEE Xplore, and Google Scholar from November 2022 to March 2026. Eligible studies reported quantitative data evaluating general-purpose LLMs for direct mental health care tasks. Methodological quality was assessed using the Mixed Methods Appraisal Tool and certainty of evidence using GRADE (Grading of Recommendations, Assessment, Development, and Evaluation) domains. Findings were narratively synthesized, and random-effects meta-analyses were conducted on studies that tested LLMs on screening and diagnosis by ChatGPT-4 (OpenAI), ChatGPT-3.5 (OpenAI), and GPT-3 (OpenAI) models for mental health outcomes that reported on specificity and sensitivity. Hartung-Knapp adjustments were applied, and prediction intervals (PIs) estimated.</p></sec><sec sec-type="results"><title>Results</title><p>We included 66 studies, comprising 37 vignette or simulation studies, 22 retrospective studies, and 7 prospective studies. Applications included screening and diagnosis (n=29), clinical decision support (n=14), treatment support (n=10), documentation and monitoring (n=6), patient education (n=4), and patient engagement (n=3). In screening and diagnosis, meta-analysis of 8 studies found that GPT-4 had the higher pooled sensitivity (0.83, 95% CI 0.38&#x2010;0.97; 95% PI 0.02&#x2010;1.00) than GPT-3.5 (0.70, 95% CI 0.13&#x2010;0.97; 95% PI 0.00&#x2010;1.00) and GPT-3 (0.61, 95% CI 0.33&#x2010;0.82; 95% PI 0.10&#x2010;0.96). However, GPT-4 specificity was lower at 0.77 (95% CI 0.52&#x2010;0.91; 95% PI 0.10&#x2010;0.99). Narrative synthesis suggested that LLMs performed most consistently in structured and linguistically explicit tasks. Certainty of evidence was generally low across domains, although documentation and monitoring reached moderate certainty. Major limitations included indirectness from vignette-based designs, uncertain representativeness, inconsistent outcome reporting, and sparse prospective real-world evaluation.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>General-purpose LLMs show promise for selected mental health care applications. However, current evidence remains too heterogeneous, indirect, and uncertain to support routine unsupervised use, particularly for diagnosis, risk assessment, crisis response, or therapeutic interaction. Broad accessibility should not be mistaken for clinical readiness. Future studies should move beyond simulations and retrospective evaluations toward prospective, real-world research assessing safety, reliability, equity, acceptability, clinical outcomes, and implementation in diverse mental health care settings.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>generative AI</kwd><kwd>mental health</kwd><kwd>primary health care</kwd><kwd>screening</kwd><kwd>clinical decision support</kwd><kwd>psychotherapy</kwd><kwd>digital health</kwd><kwd>eHealth</kwd><kwd>PRISMA</kwd><kwd>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Mental health conditions are a major contributor to global disease burden [<xref ref-type="bibr" rid="ref1">1</xref>]; yet, access to timely and appropriate care remains limited by persistent shortages in the mental health workforce [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. This is particularly concerning in low-resource settings, where demand for mental health support often substantially exceeds available service capacity. As health systems seek scalable ways to expand access, recent advances in generative large language models (LLMs; eg, ChatGPT [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]) have attracted growing interest as tools that may assist selected aspects of mental health care delivery.</p><p>Mental health care is a particularly important setting for the evaluation of generative LLMs. For mental health care, spoken and written language underpin diagnostic assessment, risk evaluation, case formulation, therapeutic alliance, and monitoring of treatment response [<xref ref-type="bibr" rid="ref7">7</xref>]. Several key symptoms of mental disorders are themselves expressed through language, including disorganized speech in psychosis and increased talkativeness in mania [<xref ref-type="bibr" rid="ref8">8</xref>]. As a result, mental health care is especially exposed to both the opportunities and risks of language-based AI systems. This creates clear potential for LLMs to support mental health care, but also important risks. Errors in interpretation, hallucinated content, limited contextual understanding, and unsafe conversational responses may directly affect clinical judgment, patient safety, and therapeutic trust [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>Research on LLMs in mental health has expanded rapidly since the public release of ChatGPT in late 2022. Previous reviews show that these systems have already been explored across a broad range of applications, including early detection and screening from text, conversational agents, psychoeducation, clinician-facing decision support, therapy-related tasks, and documentation-related uses [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. However, these reviews also indicate that the field remains at an early stage and that much of the existing evidence is not yet well suited to informing clinical implementation. Many studies have relied on prompt-based experiments, vignettes, simulated conversations, cross-sectional comparisons, or evaluations of chatbot responses, with fewer studies using real patient care data or prospective designs embedded in clinical settings [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. More broadly, reviews of LLMs in health care have similarly found that evaluations have focused heavily on question answering and accuracy, with limited use of real patient care data, substantial heterogeneity in tasks and outcomes, and inconsistent evaluation methods [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>The rapid evolution of LLM technology further strengthens the rationale for reassessing the evidence base. Earlier reviews identified several limitations in the use of LLMs for mental health care, including inconsistent performance across extended interactions due to limited memory, limited clinical relevance of single-turn question-answer evaluations, and broader concerns regarding transparency and explainability [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. Since then, newer generations of LLMs have introduced much larger memory capacity [<xref ref-type="bibr" rid="ref17">17</xref>], stronger multiturn interaction capacity, and reasoning-oriented variants that can produce more explicit stepwise responses (eg, OpenAI&#x2019;s o1 [<xref ref-type="bibr" rid="ref18">18</xref>]). As such, some limitations identified in earlier studies may now be less pronounced, but the updated model may pose new risks. Consequently, conclusions based largely on older generations of models may not fully capture the capabilities and risks of newer LLM systems.</p></sec><sec id="s1-2"><title>Study Aims</title><p>This systematic review aimed to synthesize empirical evidence on the use of general-purpose LLMs in mental health care. Specifically, we aimed to identify the mental health care tasks for which these models have been evaluated; characterize the study designs, data sources, models, prompting approaches, comparators, mental health conditions, and outcome measures used; synthesize findings across clinically meaningful task domains; assess methodological quality and certainty of evidence; and consider implications for clinical practice and future research. Where groups of studies were sufficiently comparable in task, model, outcome domain, and reported performance metrics, we conducted quantitative synthesis. We focused on general-purpose LLMs because they are widely accessible and therefore among the models most likely to be used ad hoc by clinicians, patients, and health systems, making their evaluation especially relevant to implementation, safety, governance, and equity.</p><p>This review adds to existing broad reviews of LLMs and generative AI in health care by focusing specifically on direct mental health care applications, where language is both the medium of care and a central source of clinical information [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. It also extends previous mental health-specific reviews, which have generally been descriptive or scoping in nature, by organizing the evidence into clinically meaningful task domains, distinguishing vignette-based, retrospective, and prospective evaluations, assessing certainty of evidence, and conducting quantitative synthesis only where sufficiently comparable evidence was available [<xref ref-type="bibr" rid="ref9">9</xref>]. This approach clarifies not only where general-purpose LLMs appear promising, but also where the evidence remains too heterogeneous, indirect, or uncertain to support routine clinical implementation.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Search Strategy</title><p>This systematic review and meta-analysis was conducted in accordance with the PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) guidelines (see <xref ref-type="supplementary-material" rid="app3">Checklist 1</xref> for PRISMA checklist; see <xref ref-type="supplementary-material" rid="app4">Checklist 2</xref> for PRISMA-S [Preferred Reporting Items for Systematic Reviews and Meta-Analyses literature search extension] checklist) [<xref ref-type="bibr" rid="ref20">20</xref>]. The review protocol was registered on PROSPERO (International Prospective Register of Systematic Reviews) [CRD420251056593], which was a broader review on AI applications in any primary health care setting. This current paper deviates from the broader PROSPERO in that we focus on clinical applications in mental health care settings. This narrower focus was adopted for the current review because of the rapid expansion of original studies in mental health and the need for a dedicated synthesis of this emerging evidence base.</p><p>We searched PubMed, Embase, ACM Digital Library, IEEE Xplore, and Google Scholar for eligible studies. We also manually searched the reference lists of relevant systematic reviews and included papers to identify additional records. The search combined terms related to LLMs (eg, &#x201C;large language model,&#x201D; &#x201C;ChatGPT,&#x201D; &#x201C;GPT,&#x201D; &#x201C;Claude,&#x201D; and &#x201C;Gemini&#x201D;) with terms related to mental health (eg, &#x201C;mental health,&#x201D; &#x201C;mental disorder,&#x201D; &#x201C;depression,&#x201D; &#x201C;anxiety,&#x201D; &#x201C;suicide,&#x201D; &#x201C;psychiatry,&#x201D; and &#x201C;psychotherapy&#x201D;). Full search strings are in Section S2 (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). For Google Scholar, results were screened in order of relevance until no further potentially eligible studies were identified.</p><p>The search was first conducted on January 11, 2025, and updated on June 26, 2025, and again on March 16, 2026. No language restrictions were applied at the search stage or in exclusion criteria, but all included studies were published in English.</p></sec><sec id="s2-2"><title>Selection Criteria</title><p>The selection criteria were developed based on population or setting, intervention or exposure, outcomes, and study types (Section S1, <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The population and setting criteria included studies conducted in primary mental health care settings or scenarios (eg, general practice, family medicine, community health centers, and integrated care) for use by individual or group users, such as patients, primary care providers, and mental health practitioners. We excluded studies on populations or settings not relevant to primary mental health care, for example, individual self-help contexts outside clinical care, social media datasets not linked to a defined clinical or decision-making context, health promotion, academic, psychiatry student examinations, and mental health research applications [<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref23">23</xref>].</p><p>For intervention or exposure, we included general-purpose LLMs, referring to large-scale, pretrained models trained broadly and designed for multipurpose use, without domain-specific training or restriction to clinical mental health applications, without fine-tuning, and without specific use of retrieval augmentation tools (including but not limited to the ChatGPT series, Claude [Anthropic PBC] series, Llama [Meta] series, PaLM/Gemini [Google LLC] series, and Mistral series). We were open to including papers that applied the general-purpose models to any mental health care setting. We excluded studies focusing only on image generation models, domain-specific or fine-tuned models, or small specialized systems developed for specific projects. Chatbot-based studies were eligible if the chatbot functioned primarily as an interface to a general-purpose LLM. Studies were excluded if they evaluated fine-tuned, custom-trained, or otherwise specialized chatbot systems whose performance was not representative of a general-purpose LLM.</p><p>We included any outcomes examined, which, based on previous reviews, included accuracy or performance metrics, time efficiency, objectively measured user satisfaction, impact on clinical decision-making, or patient mental health outcomes. Studies without outcomes quantified were excluded.</p><p>Study types included were empirical human studies with quantitative data, such as experimental, observational, implementation, evaluation, or mixed-methods studies with measurable outcomes. Opinion pieces, commentaries, reviews, theoretical frameworks, and purely qualitative studies without quantified outcomes were excluded. Preprints, conference papers, and abstracts were included if they met all the criteria, but those with no quantitative data were excluded.</p><p>We included studies published from November 2022 onward, corresponding to the public release and widespread accessibility of ChatGPT, the first widely accessible and publicly available general-purpose LLM, in any language and publication type (including peer-reviewed papers, preprints, and conference papers).</p></sec><sec id="s2-3"><title>Study Selection</title><p>All records identified through database and supplementary searches were imported into Covidence for deduplication and screening. Title and abstract screening and full-text screening were conducted independently by two reviewers. Interreviewer agreement across screening decisions was 70.5% (10,957/15,542). Disagreements were resolved through discussion, and a third reviewer was consulted when consensus could not be reached. Screening decisions were guided by the predefined eligibility criteria, and common reasons for disagreement were discussed within the review team to support consistent interpretation of the criteria.</p><p>To minimize double counting, we examined studies with overlapping authors, datasets, or study designs and descriptions for possible duplication. Where multiple reports appeared to use the same or partially overlapping samples, input data, or outcomes, duplicate reports were excluded, and the most relevant or complete report was retained. Studies based on the same datasets were retained if different LLMs were tested or if there were different mental health outcomes.</p></sec><sec id="s2-4"><title>Data Extraction</title><p>A standardized data extraction form was developed and piloted on 3 included studies before formal extraction commenced. Data were extracted by one reviewer and independently checked by a second reviewer. Any discrepancies were resolved through discussion, with consultation from a third reviewer when needed.</p><p>The extracted information included study characteristics (eg, country or region, study design, clinical setting or scenario, data source or population, and target mental health condition), intervention characteristics (eg, model name, model type, prompting approach, and primary mental health care task), and evaluation characteristics (eg, comparator or reference standard, outcome measures, and key findings).</p></sec><sec id="s2-5"><title>Data Analysis</title><p>Methodological quality and risk of bias were appraised using the Mixed Methods Appraisal Tool (MMAT), which was selected because it provides design-specific criteria applicable to different study designs [<xref ref-type="bibr" rid="ref24">24</xref>]. MMAT ratings were completed by 1 reviewer and independently checked by a second reviewer. Any disagreements were resolved through discussion until consensus was reached.</p><p>A narrative synthesis was used to summarize findings across the full set of included studies because substantial heterogeneity was anticipated in study design, clinical context, model type, prompting approach, and outcome measurement. For synthesis, studies were grouped according to the primary mental health care task evaluated based on studies identified, which were screening and diagnosis, clinical decision support, treatment support, documentation and monitoring, patient education, and patient engagement. These groupings were developed post hoc based on the characteristics of the included studies and were used to support structured comparison across task domains. Similarly, we also grouped the studies based on what designs the included papers had used (eg, prospective, retrospective, or vignette-based design). Prospective studies involved real-time data collection or live interaction with LLMs, including both real-world implementation and controlled research interview settings. Retrospective studies analyzed or tested LLMs on preexisting datasets without live deployment. Vignette-based studies evaluated LLMs using constructed scenarios or standardized prompts rather than real-world data.</p><p>Whether to conduct quantitative synthesis and meta-analyses was based on criteria used to determine if comparable estimates were available for pooling. This reflected both clinical and statistical considerations. Studies were grouped according to (1) mental health care task type; (2) LLM model tested; (3) mental disorder, symptom domain, or clinical outcome examined, guided where applicable by <italic>DSM</italic> (<italic>Diagnostic and Statistical Manual of Mental Disorders</italic>) or <italic>ICD</italic> (<italic>International Classification of Diseases</italic>) classifications; and (4) the quantitative metrics reported. Meta-analysis was conducted only when at least four estimates were available within the same task group [<xref ref-type="bibr" rid="ref25">25</xref>].</p><p>Based on these criteria, meta-analysis was restricted to studies in the screening and diagnosis task group that reported sensitivity and specificity, or provided sufficient information to derive these estimates. As too few studies were available to conduct disorder-specific meta-analyses, we grouped studies within a broader internalizing distress and suicide-risk outcome family. This grouping included depressive and anxiety disorders, posttraumatic stress disorder (PTSD)&#x2013;related outcomes, and suicidality-related outcomes. These outcomes were not treated as equivalent diagnoses; rather, they were grouped because they represent clinically overlapping mental health presentations commonly assessed in screening and risk-detection contexts.</p><p>As the outcomes to be pooled were sensitivity and specificity estimates, we conducted random-effects bivariate diagnostic test accuracy meta-analyses using the <italic>MIDAS</italic> package in Stata/SE 18 (StataCorp LLC). A bivariate approach was chosen because sensitivity and specificity are correlated and may vary jointly across studies, so modeling both outcomes together in the same model is appropriate. Separate meta-analyses were conducted by LLM models with at least four estimates for pooling, which included GPT-4, GPT-3.5, and GPT-3. Other models did not have sufficiently comparable data for pooling. It should be noted that in this review, we used the term &#x201C;GPT-version&#x201D; instead of referring to the generative pretrained transformer (GPT) family model generically as ChatGPT because we considered ChatGPT to be a web interface that allows connection to different versions of the model. We use the term ChatGPT without the version number when the original study authors did not report the version of GPT used. We did not calculate an overall pooled estimate across all models because combining different LLMs into a single summary estimate was not considered meaningful. We planned to conduct sensitivity analyses by excluding outliers, but there were no outliers identified; therefore, the sensitivity analyses were not conducted. From these models, we estimated pooled sensitivity, specificity, positive likelihood ratio, negative likelihood ratio, and diagnostic odds ratio with 95% CIs. Between-study heterogeneity was assessed using Cochran Q and <italic>I</italic>&#x00B2; statistics, and model adequacy was examined using diagnostic plots [<xref ref-type="bibr" rid="ref26">26</xref>]. We estimated prediction intervals (PIs) and CIs, and Hartung-Knapp adjustments were applied [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>].</p><p>In addition, we used evidence from the MMAT, narrative review, and meta-analyses to determine the current level of evidence for each of the mental health care tasks that LLMs had been tested on, based on the GRADE (Grading of Recommendations, Assessment, Development, and Evaluation) domains [<xref ref-type="bibr" rid="ref29">29</xref>].</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Selection</title><p>We identified a total of 28,893 records through database and register searching (PubMed n=11,091; ACM Digital Library n=815; IEEE Xplore n=2302; Embase n=857; Google Scholar n=13,828). After removing 12,777 duplicate records and 979 records for other reasons, 15,137 records were screened, of which 14,732 were excluded at title and abstract screening. We then screened 405 full-texts for eligibility. Finally, a total of 66 studies met the inclusion criteria and were included in the systematic review, with 8 studies included in the meta-analysis (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flow diagram of study selection.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e87730_fig01.png"/></fig></sec><sec id="s3-2"><title>Study Characteristics</title><p>The characteristics of the 66 included studies are presented in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. Across the included studies, 48.5% (32/66) of studies were conducted in North America, 15.2% (10/66) of studies in Europe, 19.7% (13/66) of studies in East Asia, 18.2% (12/66) of studies in the Middle East, 4.5% (3/66) of studies in South Asia, and 3% (2/66) of studies in Oceania; some studies spanned multiple regions.</p><p>By study design, 56.1% (37/66) of studies were vignette or simulation studies, 33.3% (22/66) of studies were retrospective studies, and 10.6% (7/66) of studies were prospective studies. By application type, 43.9% (29/66) of studies focused on screening and diagnosis, 21.2% (14/66) of studies on clinical decision support, 15.2% (10/66) of studies on treatment support, 9.1% (6/66) of studies on documentation and monitoring, 6.1% (4/66) of studies on patient education, and 4.5% (3/66) of studies on patient engagement.</p><p>Target conditions were most commonly general or nonspecific mental health presentations, examined in 31.8% (21/66) of studies, followed by depression in 28.8% (19/66) of studies and suicidality or suicide risk in 16.7% (11/66) of studies. Anxiety-related conditions and schizophrenia or psychosis were each examined in 7.6% (5/66) of studies, bipolar or other mood disorders in 6.1% (4/66) of studies, and PTSD in 4.5% (3/66) of studies. Autism spectrum disorder, obsessive-compulsive disorder (OCD), and psychiatric emergencies or acute psychiatric crises were each examined in 3% (2/66) of studies. Attention-deficit/hyperactivity disorder and substance use or addiction psychiatry were each examined in 1.5% (1/66) of studies. As some studies addressed more than one target condition, these categories were not mutually exclusive.</p></sec><sec id="s3-3"><title>Quality Assessment</title><p>Based on the MMAT appraisal, methodological quality was generally acceptable, although study quality varied across designs (Section S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Among the quantitative nonrandomized studies, 60.4% (29/48) met at least 4 of the 5 MMAT criteria. The most common limitation was sample representativeness: this criterion was rated as unclear in 31.3% (15/48) of studies and not met in 12.5% (6/48) of studies, suggesting potential limitations in generalizability. Control for confounding was reported in 31.3% (15/48) of studies. Among the quantitative descriptive studies, 75% (6/8) of studies met at least 4 of the 5 MMAT criteria, with the main concerns relating to sample representativeness and nonresponse bias. Among the mixed methods studies, 90% (9/10) met at least 4 of the 5 MMAT criteria. Where mixed methods criteria were not fully met, this was usually due to limited justification for the mixed methods design, insufficient integration of qualitative and quantitative findings, or inadequate discussion of divergences between components. Overall, the evidence base was methodologically heterogeneous, and the principal concern was uncertainty regarding representativeness rather than pervasive major flaws in study conduct. The full assessment can be seen in Section S5 (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). In the following section, studies are summarized by application types.</p></sec><sec id="s3-4"><title>Screening and Diagnosis</title><p>Screening and diagnosis referred to tasks in which LLMs were used to classify or identify a mental health condition or estimate symptom risk or severity. We identified 29 studies on screening and diagnosis, including 14 retrospective, 13 vignette, and 2 prospective studies.</p><p>The 29 studies used heterogeneous methods to evaluate LLM performance in mental health screening and diagnostic classification, including retrospective analyses of patient narrative [<xref ref-type="bibr" rid="ref30">30</xref>], clinical interview transcripts [<xref ref-type="bibr" rid="ref31">31</xref>-<xref ref-type="bibr" rid="ref36">36</xref>], diary entries [<xref ref-type="bibr" rid="ref37">37</xref>], hospital clinical notes and discharge summaries [<xref ref-type="bibr" rid="ref38">38</xref>-<xref ref-type="bibr" rid="ref40">40</xref>], sentence completion narratives [<xref ref-type="bibr" rid="ref41">41</xref>], inpatient psychiatric admission narratives [<xref ref-type="bibr" rid="ref42">42</xref>], and structured questionnaire responses [<xref ref-type="bibr" rid="ref43">43</xref>]. In addition, prospective studies used LLM-derived sentiment ratings from brief written responses [<xref ref-type="bibr" rid="ref44">44</xref>] and depression screening through conversations [<xref ref-type="bibr" rid="ref45">45</xref>]. The remaining studies were vignette-based or other simulated evaluations, including <italic>DSM-5</italic> (<italic>Diagnostic and Statistical Manual of Mental Disorders</italic> [Fifth Edition]) or <italic>DSM-5-TR</italic> (<italic>Diagnostic and Statistical Manual of Mental Disorders</italic> [Fifth Edition, Text Revision]) psychiatric diagnostic vignettes [<xref ref-type="bibr" rid="ref46">46</xref>-<xref ref-type="bibr" rid="ref48">48</xref>], suicidality and suicide-risk scenarios [<xref ref-type="bibr" rid="ref48">48</xref>-<xref ref-type="bibr" rid="ref53">53</xref>], childhood anxiety vignettes [<xref ref-type="bibr" rid="ref54">54</xref>], and OCD diagnostic vignettes [<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>]. Reference standards and comparators were similarly diverse, including validated symptom scales and thresholds [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref43">43</xref>], psychiatrist adjudication or clinical diagnosis [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref45">45</xref>], human annotation with physician validation [<xref ref-type="bibr" rid="ref38">38</xref>], benchmark <italic>DSM</italic>-based vignette diagnoses [<xref ref-type="bibr" rid="ref46">46</xref>-<xref ref-type="bibr" rid="ref48">48</xref>], and mental health professional comparators [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref58">58</xref>].</p><p>Across the vignette-based screening and diagnosis studies, LLM performance was mixed. Stronger results were reported in more narrowly defined vignette tasks, such as childhood anxiety disorders, where LLMs often matched or exceeded human comparators or achieved high diagnostic accuracy [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref57">57</xref>]. Performance was also strong in OCD studies, where LLMs achieved high diagnostic accuracy and outperformed human comparators [<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>]. In broader psychiatric vignette studies, GPT-3.5 produced broadly acceptable responses across 100 psychiatric case vignettes [<xref ref-type="bibr" rid="ref59">59</xref>], while newer frontier models showed moderate-to-strong diagnostic performance, with better reasoning associated with greater diagnostic accuracy [<xref ref-type="bibr" rid="ref60">60</xref>]. Similarly, in a broader psychiatric vignette set, performance was high for depression, social phobia, and PTSD, but weaker for schizophrenia [<xref ref-type="bibr" rid="ref58">58</xref>].</p><p>Structured approaches also improved performance in some cases, such as self-verification prompting, which increased positive predictive value in <italic>DSM-5-TR</italic> case diagnosis in a study [<xref ref-type="bibr" rid="ref48">48</xref>], but reduced sensitivity [<xref ref-type="bibr" rid="ref61">61</xref>]. However, important limitations were mentioned in some studies, including weaker performance for disorders in which the diagnosis depends on the symptoms as well as the timing and course of the symptom presentations, such as peripartum depression and acute stress disorder [<xref ref-type="bibr" rid="ref47">47</xref>]. In addition, limitations mentioned included variability across diagnostic categories and demographic bias [<xref ref-type="bibr" rid="ref46">46</xref>], delayed escalation and limited crisis referral in simulated suicidality scenarios [<xref ref-type="bibr" rid="ref49">49</xref>], and systematic deviation from professional norms in suicide-risk judgments, with risks over- or underestimated depending on model type [<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>]. Other vignette-based studies further highlighted variability in referral decisions, intervention stability, and benchmark performance across broader psychiatric or suicide-related tasks [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref62">62</xref>].</p><p>Across the retrospective studies, findings were similarly mixed. LLMs performed well for autism spectrum disorder, depression, and PTSD estimation from interviews and inpatient psychiatric diagnosis from admission narratives, although conventional models still outperformed LLMs in some settings [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref63">63</xref>]. Some studies also reported promising performance for symptom scoring from interview transcripts and suicide-risk estimation from discharge summaries, although discrimination for suicide-risk prediction remained only modest [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. A retrospective study also showed strong performance for binary condition identification from hospital clinical notes, as illustrated by high sensitivity and specificity for depression extraction in 1 small emergency health record study [<xref ref-type="bibr" rid="ref38">38</xref>]. However, performance was weaker in other applications, including childbirth-related PTSD screening, where sensitivity was low despite high specificity, and adolescent suicidal-ideation estimation, where traditional machine-learning models slightly outperformed LLM-based approaches [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref43">43</xref>]. Other retrospective studies also showed only moderate performance for depression and suicide-risk prediction from sentence-completion narratives and for extraction of mental health causes, although prompting refinements improved results in some settings [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref64">64</xref>].</p><p>In the 2 prospective studies, a GPT-based fast screen showed high agreement with clinical diagnosis and outperformed the validated Patient Health Questionnaire scale for screening for depression, but in a very small sample [<xref ref-type="bibr" rid="ref45">45</xref>]. In a separate prospective study, LLM-derived sentiment ratings from brief written responses were associated with both current and future depression severity and significantly predicted short-term worsening, suggesting potential utility for language-based mood monitoring rather than direct diagnosis [<xref ref-type="bibr" rid="ref44">44</xref>].</p><p>For consideration in the meta-analysis, of the 29 studies on screening and diagnosis summarized above, 8 provided sufficiently comparable data on depression and anxiety disorders, PTSD-related outcomes, and suicidality-related outcomes included [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref46">46</xref>]. Reference standards varied across studies and included validated symptom scales [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref43">43</xref>], symptom scale thresholds with psychiatrist adjudication [<xref ref-type="bibr" rid="ref37">37</xref>], psychiatrist clinical diagnosis [<xref ref-type="bibr" rid="ref45">45</xref>], manual human annotation with physician validation [<xref ref-type="bibr" rid="ref38">38</xref>], and <italic>DSM-5</italic>-based vignette diagnoses [<xref ref-type="bibr" rid="ref46">46</xref>].</p><p>Visual inspection of the meta-analyses&#x2019; diagnostic outputs did not suggest concerns about the model fit, and funnel plot asymmetry tests were not statistically significant (GPT-4 <italic>P</italic>=.79, GPT-3.5 <italic>P</italic>=.31, GPT-3 <italic>P</italic>=.42), noting the small number of studies (<xref ref-type="table" rid="table1">Table 1</xref>; Sections S2-S4, <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The bivariate meta-analysis found that GPT-4 had the highest pooled sensitivity (0.83, 95% CI 0.38&#x2010;0.97; 95% PI 0.02&#x2010;1.00), but lower specificity (0.77, 95% CI 0.52&#x2010;0.91; 95% PI 0.10&#x2010;0.99) than GPT-3.5 and GPT-3, but with large CIs and PIs. GPT-3.5 showed a pooled sensitivity of 0.70 (95% CI 0.13&#x2010;0.97; 95% PI 0.00&#x2010;1.00) and specificity of 0.96 (95% CI 0.95&#x2010;0.96; 95% PI 0.95&#x2010;0.96), while GPT-3 showed a pooled sensitivity of 0.61 (95% CI 0.33&#x2010;0.82; 95% PI 0.10&#x2010;0.96) and specificity of 0.94 (95% CI 0.88&#x2010;0.96; 95% PI 0.87&#x2010;0.96). GPT-3.5 had the highest positive likelihood ratio (15.90, 95% CI 9.30&#x2010;27.30), whereas GPT-4 had the lowest negative likelihood ratio (0.23, 95% CI 0.11&#x2010;0.47). Heterogeneity was large. Overall, the meta-analyses showed that GPT-4 had the highest pooled sensitivity, but poor specificity with wide adjusted CIs, and wide PIs indicating substantial between-study variability in expected performance across settings.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Bivariate meta-analysis of diagnostic test accuracy of LLMs<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> on internalizing and distress-related mental conditions. There were not enough like-for-like studies for meta-analyses for other LLMs, other areas of mental health care applications, and other mental health outcomes. Forest plots and diagnostic plots are available in Sections S2-S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">GPT-4 (k=4)</td><td align="left" valign="bottom">GPT-3.5 (k=4)</td><td align="left" valign="bottom">GPT-3 (k=5)</td></tr></thead><tbody><tr><td align="left" valign="top">Pooled sensitivity, pooled estimate (95% CI)</td><td align="left" valign="top">0.83 (0.38-0.97)</td><td align="left" valign="top">0.70 (0.13-0.97)</td><td align="left" valign="top">0.61 (0.33-0.82)</td></tr><tr><td align="left" valign="top">Prediction intervals</td><td align="left" valign="top">0.02-1.00</td><td align="left" valign="top">0.00-1.00</td><td align="left" valign="top">0.10-0.96</td></tr><tr><td align="left" valign="top">Q for sensitivity</td><td align="left" valign="top">35.12<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">41.44<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">12.25<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">I-sq for sensitivity, pooled estimate (95% CI)</td><td align="left" valign="top">91.46 (84.75-98.16)</td><td align="left" valign="top">92.76 (87.34&#x2010;98.19)</td><td align="left" valign="top">67.36 (36.28&#x2010;98.43)</td></tr><tr><td align="left" valign="top">Specificity, pooled estimate (95% CI)</td><td align="left" valign="top">0.77 (0.52-0.91)</td><td align="left" valign="top">0.96 (0.95&#x2010;0.96)</td><td align="left" valign="top">0.94 (0.88&#x2010;0.96)</td></tr><tr><td align="left" valign="top">Prediction intervals</td><td align="left" valign="top">0.10-0.99</td><td align="left" valign="top">0.95-0.96</td><td align="left" valign="top">0.87-0.96</td></tr><tr><td align="left" valign="top">Q for specificity</td><td align="left" valign="top">58.10<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">16.50<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">6.23<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">I-sq for specificity, pooled estimate (95% CI)</td><td align="left" valign="top">94.84 (91.33-98.35)</td><td align="left" valign="top">81.82 (64.51&#x2010;99.13)</td><td align="left" valign="top">35.80 (0.00&#x2010;98.86)</td></tr><tr><td align="left" valign="top">Positive likelihood ratio, pooled estimate (95% CI)</td><td align="left" valign="top">3.60 (2.50-5.30)</td><td align="left" valign="top">15.90 (9.30-27.30)</td><td align="left" valign="top">10.50 (6.20-17.90)</td></tr><tr><td align="left" valign="top">Negative likelihood ratio, pooled estimate (95% CI)</td><td align="left" valign="top">0.23 (0.11-0.47)</td><td align="left" valign="top">0.31 (0.11-0.91)</td><td align="left" valign="top">0.42 (0.25-0.70)</td></tr><tr><td align="left" valign="top">Asymmetry test (<italic>P</italic> value)</td><td align="left" valign="top">.79</td><td align="left" valign="top">.31</td><td align="left" valign="top">.42</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table1fn2"><p><sup>b</sup><italic>P</italic>&#x003C;.01.</p></fn><fn id="table1fn3"><p><sup>c</sup><italic>P</italic>=.02.</p></fn><fn id="table1fn4"><p><sup>d</sup><italic>P</italic>=.18.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Clinical Decision Support</title><p>Clinical decision support studies included those that tested LLMs in supporting mental health-related clinical judgments, such as prognosis, triage, medication support, or evaluation of therapeutic responses. We identified 14 studies, including 13 vignette-based studies and 1 prospective simulated training study.</p><p>The clinical decision support studies covered prognosis, treatment selection, medication support, suicide-response scoring, emergency triage, psychotherapy case-response tasks, counseling benchmarks, and cognitive behavioral therapy (CBT) knowledge and therapist-response tasks. In studies examining schizophrenia and depression vignettes, models generally recommended treatment appropriately, but prognostic judgments varied across models, with GPT-3.5 tending to be more pessimistic than newer models and human comparators [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref65">65</xref>].</p><p>Studies of clinical decision support showed promise but also important safety limitations. GPT-4 often selected appropriate antidepressant treatments, but also included contraindicated or poor options in many evaluations [<xref ref-type="bibr" rid="ref66">66</xref>]. Guideline augmentation improved bipolar treatment selection substantially, although some inappropriate treatments remained [<xref ref-type="bibr" rid="ref67">67</xref>]. Retrieval-augmented psychiatric medication support also performed well, especially with GPT-4o-based systems [<xref ref-type="bibr" rid="ref68">68</xref>], and <italic>ICD-11</italic> (<italic>International Classification of Diseases, 11th Revision</italic>) criteria could be translated into highly accurate executable logic after expert correction [<xref ref-type="bibr" rid="ref55">55</xref>].</p><p>Performance was more mixed on complex response-generation tasks. Models showed bias when scoring suicide-intervention responses [<xref ref-type="bibr" rid="ref69">69</xref>], no model achieved consistently acceptable performance on psychotherapy case-response tasks [<xref ref-type="bibr" rid="ref70">70</xref>], and LLMs underperformed human reference responses on CBT therapist-response generation despite strong knowledge performance [<xref ref-type="bibr" rid="ref62">62</xref>]. In psychiatric emergency triage, GPT-4 models showed substantial agreement with clinicians, but some false positives suggested over-triage [<xref ref-type="bibr" rid="ref71">71</xref>]. Overall, these findings suggest that LLMs have considerable potential for structured mental health clinical decision support, but still require careful oversight for prognosis, safety-sensitive decisions, prescribing, and therapeutically nuanced tasks.</p></sec><sec id="s3-6"><title>Treatment Support</title><p>Treatment support referred to tasks in which LLMs were used to deliver, simulate, or evaluate therapeutic or supportive interactions, such as psychotherapy-style conversations, CBT-based responses, journaling support, or feedback on helping skills, rather than to screen for a condition or make a formal clinical decision. We identified 10 studies on treatment support, including 5 vignette-based studies, 1 retrospective study, and 4 prospective studies.</p><p>The treatment support studies examined a range of applications, including simulated CBT sessions [<xref ref-type="bibr" rid="ref72">72</xref>], anxiety support through repeated chatbot conversations [<xref ref-type="bibr" rid="ref73">73</xref>], AI-generated feedback for suicide-prevention role-play training [<xref ref-type="bibr" rid="ref74">74</xref>], CBT-style cognitive distortion and reframing tasks [<xref ref-type="bibr" rid="ref75">75</xref>], AI-supported journaling for mental well-being [<xref ref-type="bibr" rid="ref76">76</xref>], depression treatment recommendations from vignette scenarios [<xref ref-type="bibr" rid="ref53">53</xref>], psychotherapy knowledge and behavioral activation tasks [<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref77">77</xref>], responses to common psychological questions [<xref ref-type="bibr" rid="ref78">78</xref>], and retrospective CBT-style dialogue generation from psychotherapy transcripts [<xref ref-type="bibr" rid="ref79">79</xref>]. Reference standards and comparators were similarly varied, including human CBT therapists [<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref75">75</xref>], primary care physicians [<xref ref-type="bibr" rid="ref53">53</xref>], psychotherapists in training [<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref77">77</xref>], a licensed clinical psychologist [<xref ref-type="bibr" rid="ref78">78</xref>], trained human psychology raters [<xref ref-type="bibr" rid="ref74">74</xref>], human psychotherapy dialogue datasets [<xref ref-type="bibr" rid="ref79">79</xref>], and user-reported evaluations without a formal comparator [<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref76">76</xref>].</p><p>Across the 5 vignette-based treatment support studies, findings were generally positive but mixed [<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref77">77</xref>,<xref ref-type="bibr" rid="ref78">78</xref>]. LLMs performed moderately to strongly on several structured treatment-support tasks, including CBT-style exercises [<xref ref-type="bibr" rid="ref75">75</xref>], behavioral activation and psychotherapy case-scenario responses [<xref ref-type="bibr" rid="ref77">77</xref>], depression treatment recommendation [<xref ref-type="bibr" rid="ref53">53</xref>], and psychological support question answering [<xref ref-type="bibr" rid="ref78">78</xref>]. However, important limitations were also identified. Human CBT therapists outperformed GPT-3.5 across multiple CBT competence domains in simulated therapy sessions [<xref ref-type="bibr" rid="ref72">72</xref>]. In depression vignettes, GPT-3.5 and GPT-4 more often recommended psychotherapy, whereas physicians more often recommended pharmacotherapy [<xref ref-type="bibr" rid="ref53">53</xref>]. Performance also varied across models, with GPT-4 outperforming GPT-3.5 on common psychological prompts in 1 study [<xref ref-type="bibr" rid="ref78">78</xref>].</p><p>In the 1 retrospective study, LLMs showed feasible CBT-style dialogue generation from psychotherapy transcripts, with stronger performance in multiturn than single-turn settings and further gains when a CBT knowledge base was added [<xref ref-type="bibr" rid="ref79">79</xref>]. Responses were generally more positive than those of human therapists, suggesting a possible overly positive bias despite reasonable dialogue quality [<xref ref-type="bibr" rid="ref79">79</xref>].</p><p>Across the 4 prospective studies, users and evaluators generally reported favorable experiences with LLM-supported interventions, including anxiety support [<xref ref-type="bibr" rid="ref73">73</xref>], AI journaling [<xref ref-type="bibr" rid="ref76">76</xref>], suicide-prevention role-play feedback [<xref ref-type="bibr" rid="ref74">74</xref>], and psychotherapy training support [<xref ref-type="bibr" rid="ref58">58</xref>]. Among adults with anxiety disorders, most participants reported that GPT-3.5 understood their anxiety accurately, and ratings of helpfulness, empathy, trustworthiness, and effectiveness were generally favorable, although privacy, ethics, and lack of human connection remained common concerns [<xref ref-type="bibr" rid="ref73">73</xref>]. AI-supported journaling also received high ratings across counseling skill, behavior, and learning domains [<xref ref-type="bibr" rid="ref76">76</xref>]. In suicide-prevention role-play training, AI-generated feedback correlated strongly with human ratings, although it tended to score performances more favorably than human raters [<xref ref-type="bibr" rid="ref74">74</xref>]. In psychotherapy training tasks, LLMs performed similarly to or better than psychotherapists in training on several knowledge and response-quality indicators [<xref ref-type="bibr" rid="ref53">53</xref>]. Overall, these findings suggest that LLMs may be useful as supportive or training-adjunct tools, but their outputs still require cautious interpretation in sensitive therapeutic contexts [<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref74">74</xref>].</p></sec><sec id="s3-7"><title>Documentation and Monitoring</title><p>Documentation and monitoring referred to tasks in which LLMs were used to summarize sessions, generate discharge summaries, standardize notes, extract structured information from free-text records, score written documentation, or monitor symptom change from routine clinical text. We identified 6 studies on documentation and monitoring, including 5 retrospective studies and 1 vignette-based study.</p><p>These 6 studies used heterogeneous methods to evaluate LLM performance in documentation and monitoring tasks, including structured summarization of counseling-session transcripts [<xref ref-type="bibr" rid="ref80">80</xref>], classification of mental health concepts from emergency electronic health record (EHR) text [<xref ref-type="bibr" rid="ref81">81</xref>], scoring of suicide safety-plan documentation [<xref ref-type="bibr" rid="ref82">82</xref>], proofreading and structured information extraction from addiction psychiatry notes [<xref ref-type="bibr" rid="ref83">83</xref>], symptom monitoring from psychiatric EHR language during clozapine treatment [<xref ref-type="bibr" rid="ref84">84</xref>], and generation of psychiatric discharge summaries from structured case information [<xref ref-type="bibr" rid="ref85">85</xref>]. Reference standards and comparators were similarly diverse, including human-annotated reference summaries [<xref ref-type="bibr" rid="ref80">80</xref>], consensus coding by expert clinicians [<xref ref-type="bibr" rid="ref81">81</xref>], trained clinical coders using a structured scoring algorithm [<xref ref-type="bibr" rid="ref82">82</xref>], human-annotated gold-standard proofread notes and extraction labels with comparison against non-LLM tools [<xref ref-type="bibr" rid="ref83">83</xref>], human-rated mental health symptom scores and conventional natural language processing features [<xref ref-type="bibr" rid="ref84">84</xref>], and human-written discharge summaries by physicians and psychotherapists [<xref ref-type="bibr" rid="ref85">85</xref>].</p><p>Across the 5 retrospective studies, findings were overall positive, with variation by task. LLMs performed well on structured summarization, concept classification, documentation scoring, information extraction, and symptom monitoring from routine clinical language [<xref ref-type="bibr" rid="ref80">80</xref>-<xref ref-type="bibr" rid="ref84">84</xref>]. In counseling-session summarization, hallucinations were uncommon across outputs [<xref ref-type="bibr" rid="ref80">80</xref>]. In emergency EHRs, where clinicians have to file cases under mental health or physical health categories, performance was strongest for the binary mental-vs-physical distinction, with &#x03BA;=0.77, precision 0.93, recall 0.93, and <italic>F</italic><sub>1</sub>-score 0.93, but dropped for finer-grained categorization across specific mental and physical health classes [<xref ref-type="bibr" rid="ref81">81</xref>]. In suicide safety-plan scoring, best <italic>F</italic><sub>1</sub>-scores ranged from 0.77 to 0.92 across sections [<xref ref-type="bibr" rid="ref82">82</xref>]. In addiction psychiatry notes, larger LLMs outperformed simpler non-LLM tools for proofreading, and GPT-4o performed strongly on structured information extraction, with a mean <italic>F</italic><sub>1</sub>-score of 0.97 for substance-class detection, an exact match of 0.96 for time since last use, and mean accuracy of 0.97 for adequacy of time-since-last-use information [<xref ref-type="bibr" rid="ref83">83</xref>]. In monitoring applications, LLM-derived psychiatric symptom severity scores captured symptom improvement during clozapine treatment and correlated positively with human-rated measures for key psychotic and behavioral features [<xref ref-type="bibr" rid="ref84">84</xref>]. Overall, performance appeared strongest for more structured extraction and classification tasks, whereas results were mixed for more specific categorization and documentation tasks, suggesting that LLMs performed more consistently in tasks with low complexity.</p><p>In the 1 vignette-based study, GPT-4 generated psychiatric discharge summaries from structured case information that were of similar quality to human-written summaries, but did so substantially faster [<xref ref-type="bibr" rid="ref85">85</xref>]. These findings suggest that LLMs may be used for documentation support to save time, although the evaluation was based on only 2 cases.</p></sec><sec id="s3-8"><title>Patient Education</title><p>Patient education included tasks in which LLMs were used to answer patient or caregiver questions, provide psychoeducational information, or respond to common mental health information needs and questions. We identified 4 studies on patient education, including 3 vignette-based studies and 1 retrospective study.</p><p>These studies examined responses to autism-related consultation questions in a real-world online care setting [<xref ref-type="bibr" rid="ref86">86</xref>], real-world mental health questions from an online counseling forum evaluated in a simulated format [<xref ref-type="bibr" rid="ref87">87</xref>], psychosis-related psychoeducation questions [<xref ref-type="bibr" rid="ref88">88</xref>], and common antidepressant-related questions paired with short patient scenarios [<xref ref-type="bibr" rid="ref89">89</xref>]. Reference standards included physician responses [<xref ref-type="bibr" rid="ref86">86</xref>], licensed mental health professional ratings together with comparison against online human therapist responses and LLM-as-judge evaluations [<xref ref-type="bibr" rid="ref87">87</xref>], psychiatrist and psychologist ratings [<xref ref-type="bibr" rid="ref88">88</xref>], and blinded psychiatrist ratings comparing LLM and psychiatrist responses [<xref ref-type="bibr" rid="ref89">89</xref>].</p><p>Across the 3 vignette-based studies, GPT-4 provided highly accurate, clear, and clinically useful psychoeducation for psychosis-related questions, although inclusivity was the weakest-rated domain and readability was only moderate [<xref ref-type="bibr" rid="ref88">88</xref>]. In responses to general mental health questions, a study showed that performance varied across models, with LLaMA-3.3 showing the strongest overall performance, while GPT-4 had lower overall quality and more refusal-style safety disclaimers, and Gemini showed lower empathy despite similar factual consistency [<xref ref-type="bibr" rid="ref87">87</xref>]. Human therapist responses scored lower overall than LLM responses in that study, although 7%&#x2010;14% of LLM outputs still contained unauthorized medical advice [<xref ref-type="bibr" rid="ref87">87</xref>]. For antidepressant-related questions, GPT-4o performed similarly to psychiatrists in accuracy, was more concise, but produced fewer clear responses, with no significant difference in readability [<xref ref-type="bibr" rid="ref89">89</xref>].</p><p>In the retrospective study, physicians were preferred overall to GPT-4 for autism-related consultation questions, with physicians scoring higher on relevance and usefulness, whereas ChatGPT scored highest on empathy and slightly exceeded physicians on correctness [<xref ref-type="bibr" rid="ref86">86</xref>]. Overall, these findings on patient education and question answering suggest that LLMs can provide useful patient education responses in mental health settings, but performance remains dependent on the model used and the evaluation domain, with ongoing concerns around clarity, safety, and response quality relative to clinicians [<xref ref-type="bibr" rid="ref86">86</xref>-<xref ref-type="bibr" rid="ref89">89</xref>].</p></sec><sec id="s3-9"><title>Patient Engagement</title><p>Patient engagement referred to tasks in which LLMs were used as interactive, user-facing systems to engage people in mental health conversations or crisis-oriented exchanges, rather than primarily to provide formal diagnosis or structured treatment. Among the 3 patient engagement studies, all 3 were conducted in simulated or vignette-based settings [<xref ref-type="bibr" rid="ref90">90</xref>-<xref ref-type="bibr" rid="ref92">92</xref>], and none used retrospective or prospective real-world clinical data.</p><p>LLMs were generally perceived as capable of providing supportive or contextually appropriate responses, particularly in lower-risk or general mental health scenarios, but important limitations were identified in more complex or crisis-sensitive situations [<xref ref-type="bibr" rid="ref69">69</xref>,<xref ref-type="bibr" rid="ref90">90</xref>,<xref ref-type="bibr" rid="ref92">92</xref>]. In a psychiatric crisis role-play study, participants rated GPT-3.5 positively overall for helpfulness, pleasantness, and appropriateness, although ratings were lower in psychosis scenarios than in depression or adjustment disorder scenarios [<xref ref-type="bibr" rid="ref90">90</xref>]. In a study of suicide-related queries, LLMs including Claude 3.5, GPT-4o, and Gemini 1.5 appropriately avoided directly answering very-high-risk questions, but showed limited ability to consistently distinguish between intermediate levels of suicide risk [<xref ref-type="bibr" rid="ref69">69</xref>]. In a comparison with licensed therapists, LLMs demonstrated some therapeutic elements such as reassurance and psychoeducation, but were judged to be more generic, overly directive, and potentially unsafe in crises [<xref ref-type="bibr" rid="ref92">92</xref>].</p></sec><sec id="s3-10"><title>Certainty of Evidence</title><p>Overall, taking into account the quality appraisal, meta-analysis findings, and narrative synthesis, the certainty of evidence was generally low across mental health care tasks (<xref ref-type="table" rid="table2">Table 2</xref>). The use of LLMs for documentation and monitoring of mental health care had the highest level of evidence, which was moderate. The certainty of evidence was downgraded for other mental health tasks, where evidence was sparse, mixed, or based on simulated designs. A substantial proportion of existing evidence was based on vignette-based studies, limiting the directness and clinical applicability of the evidence. Many other studies were retrospective, which was moderate, but a higher level of evidence would be achieved by having prospective or trial-based studies conducted. Across domains, there were important methodological limitations, including unclear confounding assessment, uncertain sample representativeness, inconsistent outcome measures, unclear publication bias, and imprecision in reported estimates. Heterogeneity was also substantial in the pooled screening and diagnosis studies, with wide CIs and PIs for some outcomes, indicating considerable uncertainty and variation across settings.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Certainty of evidence based on GRADE<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> assessments.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Mental health care task</td><td align="left" valign="bottom">Studies conducted</td><td align="left" valign="bottom">Summary</td><td align="left" valign="bottom">Certainty of the evidence</td></tr></thead><tbody><tr><td align="left" valign="top">Screening and diagnosis</td><td align="left" valign="top">29 studies;<break/>13 vignette,<break/>14 retrospective,<break/>2 prospective</td><td align="left" valign="top">There was a relatively larger body of evidence available, but findings were mixed across mental health outcomes and task types. Most studies relied on vignette-based or retrospective datasets, with relatively few prospective evaluations. Performance was often promising for some structured classification tasks, but results varied by condition, prompting strategy, and model, and several studies identified important safety or bias-related concerns.</td><td align="left" valign="top">Low</td></tr><tr><td align="left" valign="top">Clinical decision support</td><td align="left" valign="top">14 studies:<break/>13 vignette,<break/>0 retrospective,<break/>1 prospective</td><td align="left" valign="top">Nearly all evidence came from simulated or vignette-based tasks, limiting direct clinical applicability. Findings varied by task and model, and several studies identified clinically important weaknesses, including biased prognostic judgments, over- or undertriage, and inclusion of contraindicated or poor treatment suggestions.</td><td align="left" valign="top">Low</td></tr><tr><td align="left" valign="top">Treatment support</td><td align="left" valign="top">10 studies:<break/>5 vignette,<break/>1 retrospective,<break/>4 prospective</td><td align="left" valign="top">Evidence suggested potential usefulness of LLMs<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> in supportive or adjunctive treatment roles, but findings varied across contexts, outcomes, and prompting strategies. Prospective studies reported positive experiences with LLMs, although concerns remained regarding privacy, human nuance, and safety in sensitive settings.</td><td align="left" valign="top">Low</td></tr><tr><td align="left" valign="top">Documentation and monitoring</td><td align="left" valign="top">6 studies:<break/>1 vignette,<break/>5 retrospective,<break/>0 prospective</td><td align="left" valign="top">Findings were more consistent in showing positive results, especially for structured extraction, classification, summarization, and documentation tasks. However, all evidence came from retrospective analyses or simulated cases, with no prospective implementation studies.</td><td align="left" valign="top">Moderate</td></tr><tr><td align="left" valign="top">Patient education</td><td align="left" valign="top">4 studies:<break/>3 vignette,<break/>1 retrospective,<break/>0 prospective</td><td align="left" valign="top">Evidence was limited to a small number of mainly simulated studies. Findings generally suggested that LLMs could produce accurate and well-rated psychoeducational or question-answering responses, although safety, clarity, readability, and appropriateness concerns remained. There was a lack of direct real-world evaluation.</td><td align="left" valign="top">Low</td></tr><tr><td align="left" valign="top">Patient engagement</td><td align="left" valign="top">3 studies:<break/>3 vignette,<break/>0 retrospective,<break/>0 prospective</td><td align="left" valign="top">Evidence was sparse and based entirely on simulated interactive scenarios. Findings suggested that LLMs could be engaging and acceptable in some contexts, but concerns remained regarding suitability in complex or crisis-related situations, and the small evidence base limits confidence.</td><td align="left" valign="top">Low</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>GRADE: Grading of Recommendations, Assessment, Development, and Evaluation.</p></fn><fn id="table2fn2"><p><sup>b</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This systematic review provides a clinically focused synthesis of empirical evidence on general-purpose LLMs in mental health care. In contrast to broad reviews of LLMs across medicine, this review focuses on a setting in which language is central to clinical assessment, risk evaluation, therapeutic interaction, monitoring, and patient engagement [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. We found that general-purpose LLMs have been evaluated across a wide range of mental health care tasks, including screening and diagnosis, clinical decision support, treatment support, documentation and monitoring, patient education, and patient engagement. The main contribution is a structured assessment of what has been tested, how it has been evaluated, where evidence is beginning to accumulate, and where the evidence remains too heterogeneous or indirect to support clinical implementation. Quantitative synthesis was feasible only for a subset of screening and diagnostic studies with comparable diagnostic test accuracy metrics; all other task domains required narrative synthesis because of substantial variation in tasks, models, data sources, comparators, outcome measures, and reporting. Overall, the findings suggest that general-purpose LLMs show promise for selected mental health care applications, but the evidence remains early, uneven, and insufficient to support routine unsupervised use.</p><p>A consistent pattern across the included studies was that general-purpose LLMs performed better in tasks that were relatively structured, constrained, and linguistically explicit [<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref80">80</xref>,<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref93">93</xref>]. This was most apparent in documentation and monitoring, where models often performed well in summarization, information extraction, concept classification, and other forms of structured text processing [<xref ref-type="bibr" rid="ref83">83</xref>]. Similar strengths were also evident in some screening, diagnostic, and decision-support tasks when the input format and expected outputs were clear. Many of these tasks, such as classification, extraction, summarization, and question answering, are well-established areas of language model research in which LLMs have shown strong performance across a wide range of domains outside of mental health care [<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref95">95</xref>]. The relative strength observed in these more bounded mental health care applications is therefore broadly consistent with the known capabilities of contemporary LLMs as general-purpose language-processing systems [<xref ref-type="bibr" rid="ref96">96</xref>].</p><p>Performance was less reliable in tasks that required nuanced clinical judgment, sustained interpersonal responsiveness, or safe management of ambiguity and risk. In screening and diagnostic applications, results varied across conditions, models, and input formats, and even when overall classification performance appeared encouraging, with better detection rates in more recent LLMs, sensitivity was often less reassuring than specificity, raising concern about missed cases [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. Further, the large PIs indicated that performance cannot be predicted across settings. In treatment support, patient education, patient engagement, and some clinical decision-support applications, LLMs often produced plausible, coherent, and well-structured responses [<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref76">76</xref>], but this did not always translate into clinically appropriate or trustworthy performance. Reported limitations included generic or overly positive therapeutic responses, inconsistent crisis escalation, weaker handling of temporally nuanced diagnoses, variable prognostic judgments, and concerns about privacy, personalization, and cultural sensitivity [<xref ref-type="bibr" rid="ref97">97</xref>,<xref ref-type="bibr" rid="ref98">98</xref>]. Similar concerns have been raised in relation to LLM-supported substance use disorder interventions, where potential benefits such as low-threshold support need to be weighed against risks relating to stigma, demographic bias, hallucinated or unsafe advice, privacy, governance, and the need for robust human oversight [<xref ref-type="bibr" rid="ref99">99</xref>]. Particularly in patient engagement and crisis-related settings, the evidence remained limited and indirect, being based entirely on simulated scenarios [<xref ref-type="bibr" rid="ref90">90</xref>-<xref ref-type="bibr" rid="ref92">92</xref>]. These findings suggest that generating plausible mental health language is not equivalent to demonstrating dependable clinical judgment in more complex or high-stakes settings.</p><p>Although this review focused on general-purpose LLMs rather than specialized or fine-tuned mental health models, the broader literature suggests a similar overall pattern. Customized ChatGPT variants and fine-tuned LLMs have shown promise in simulated counseling interactions and in structured tasks such as summarizing counseling notes, discharge summaries, and medication logs [<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref80">80</xref>,<xref ref-type="bibr" rid="ref85">85</xref>]. However, these models also continued to raise concerns regarding privacy, ethics, intervention depth, and missed emotional nuance, limiting their reliability as standalone tools [<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref80">80</xref>]. Taken together, these findings suggest that both general-purpose and more specialized LLMs share similar limitations in mental health care settings [<xref ref-type="bibr" rid="ref100">100</xref>].</p><p>Our findings are broadly aligned with earlier reviews showing that LLM research in mental health has grown rapidly but has been dominated by early-stage and indirect evaluations [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Prior mental health reviews found that much of the literature consisted of prompt experiments, vignette studies, or evaluations of chatbot responses, with very few prospective studies involving participants [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Broader health care reviews reached similar conclusions, reporting that only a small proportion of studies used real patient care data and that most evaluations relied on examination questions, vignettes, or expert-generated prompts [<xref ref-type="bibr" rid="ref101">101</xref>]. Further, ethical concerns raised included fairness, bias, being too positive, transparency, and privacy [<xref ref-type="bibr" rid="ref100">100</xref>]. Compared with these earlier syntheses, the present review suggests an important, although still incomplete, progress in the evidence base. Retrospective clinical data and some prospective real-world evaluations are now beginning to appear in mental health applications of general-purpose LLMs [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref63">63</xref>]. This review also extends prior work by quantitatively synthesizing a subset of screening and diagnostic studies, whereas many earlier reviews were primarily descriptive, narrative, or scoping in nature [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. Our focus on general-purpose LLMs is also important because these are the most widely accessible models, and therefore the ones most accessible in practice-like settings or used ad hoc by clinicians and patients [<xref ref-type="bibr" rid="ref102">102</xref>]. This brings practical value to the field by clarifying not only where these models appear most promising, but also where the evidence remains too uncertain, heterogeneous, unpredictable, or indirect to support routine use.</p></sec><sec id="s4-2"><title>Limitations</title><p>This review should be interpreted in light of several limitations in the available evidence base and synthesis. First, much of the literature was still based on vignette, simulated, or retrospective studies rather than prospective evaluations embedded in routine care, as mentioned above [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref63">63</xref>]. Although these designs are useful for early-stage assessment, they do not fully capture the complexity, uncertainty, and longitudinal nature of real mental health care encounters, in which presentations are more heterogeneous, and interactions unfold over time [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. As a result, ecological validity remains limited, and the safety, clinician trust, and real-world uptake of these systems remain insufficiently tested.</p><p>Second, the evidence base was highly heterogeneous, with substantial variation in clinical tasks applied, target mental health conditions, LLM model versions, prompting strategies, comparators, data sources, and outcome measures. This limited direct comparison across studies, constrained generalizability, and reduced the extent to which findings could be synthesized quantitatively. These issues also contributed to reduced confidence in the certainty of evidence. There were still only a small number of studies that were based on human data. Future studies will need better causal design, such as well-designed cohort studies with comprehensive confounding adjustment, to estimate the effect of LLMs on various clinical tasks [<xref ref-type="bibr" rid="ref103">103</xref>,<xref ref-type="bibr" rid="ref104">104</xref>].</p><p>Third, the meta-analysis has important limitations. Quantitative synthesis was possible only after identifying groups of studies that were sufficiently comparable in task, model, outcome domain, and reported performance metric. In practice, this meant that the meta-analysis was restricted to a small number of screening and diagnostic classification studies reporting sensitivity and specificity. The included studies were not sufficiently numerous to support robust subgroup analyses by individual mental disorder, data source, study design, prompting approach, or reference standard. As a result, the pooled estimates should not be interpreted as disorder-specific diagnostic accuracy estimates. The inclusion of related but distinct outcomes, such as depression, anxiety, PTSD-related outcomes, and suicidality-related outcomes, was a pragmatic decision made because disorder-specific pooling was not feasible. Although these outcomes are clinically related and commonly evaluated in screening and risk-detection contexts, they are not interchangeable diagnoses. The wide CIs, wide PIs, and substantial heterogeneity indicate that model performance is likely to vary considerably across settings. The meta-analysis should therefore be viewed as an early and cautious synthesis of the most comparable available evidence, not as definitive evidence of clinical effectiveness or implementation readiness, and also quantifies the lack of data and uncertainty in the evidence.</p><p>Fourth, important reporting gaps limited interpretation. Prompting was often incompletely described, even though prompt design can substantially influence model outputs [<xref ref-type="bibr" rid="ref105">105</xref>]. Without transparent reporting of prompts and prompt refinement, it is difficult to determine whether observed performance differences reflected the model itself, the way it was instructed, or both. Some studies also did not adequately describe recruitment strategies or sampling frames, making it difficult to assess whether included participants or datasets were representative of the populations of interest. In addition, individuals with more complex presentations may have been underrepresented, which may have inflated estimates of model performance [<xref ref-type="bibr" rid="ref106">106</xref>].</p><p>Fifth, most studies were conducted in English, even though LLM performance may differ in non-English settings [<xref ref-type="bibr" rid="ref107">107</xref>]. This is particularly important because many non-English-speaking settings, including low- and middle-income countries, face greater shortages in mental health care resources and may have the most to gain from scalable digital tools [<xref ref-type="bibr" rid="ref108">108</xref>]. The limited linguistic and geographic diversity of the evidence therefore constrains the extent to which these findings can be assumed to apply across settings.</p><p>Finally, the evidence base is evolving rapidly. Only a limited number of studies evaluated the most recent model variants, including newer GPT-4 and GPT-4o [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref69">69</xref>,<xref ref-type="bibr" rid="ref71">71</xref>]. Accordingly, the conclusions of this review would need updating as further research becomes available, and AI-assisted living systematic review approaches may help keep pace with the rapid expansion of this literature [<xref ref-type="bibr" rid="ref109">109</xref>].</p></sec><sec id="s4-3"><title>Conclusions</title><p>This review provides a clinically focused synthesis of general-purpose LLMs in mental health care. The evidence suggests that these models are most promising for structured, language-based tasks such as documentation, summarization, information extraction, and monitoring, where the input and expected output are relatively constrained. In contrast, evidence remains much less convincing for high-stakes tasks that require nuanced clinical judgment, including diagnosis, risk assessment, prognosis, crisis response, and therapeutic interaction.</p><p>The central implication is that broad accessibility should not be mistaken for clinical readiness. General-purpose LLMs may become useful adjunctive tools in mental health care, particularly where they support clinicians rather than replace them. However, current evidence remains too heterogeneous, indirect, and uncertain to justify routine unsupervised use. Future research needs to move beyond vignette-based and retrospective evaluations toward prospective, real-world studies that assess safety, reliability, equity, acceptability, clinical outcomes, and implementation in diverse care settings.</p></sec></sec></body><back><ack><p>Generative AI was used solely for basic spelling and language checking. ChatGPT was used for limited editorial and formatting assistance during manuscript preparation. Its use was restricted to table-formatting support, including concatenation of cells, and language suggestions to improve flow and help identify possible drafting errors. The authors remain fully responsible for this paper, wrote the substantive content, critically reviewed and verified all AI-generated suggestions, and made all final decisions regarding content, interpretation, wording, citations, and revisions. All authors reviewed and edited the manuscript and take full responsibility for its content.</p></ack><notes><sec><title>Funding</title><p>This work was supported by Winter Research Programs, which provided funding for student research assistance. JL, BJ, and GCKC were supported by the Australian National Health and Medical Research Council. SF, KM, PCG, YMW, and RL were supported by the University of Queensland Research and Training Program scholarship, which provided funding for student research assistance. The funders had no role in study design, data collection, analysis, interpretation, writing, or publication decisions.</p></sec><sec><title>Data Availability</title><p>All data associated with this study are present in this paper or the supplementary information. The corresponding author can be contacted for any additional data.</p></sec></notes><fn-group><fn fn-type="con"><p>JL and GCKC conceptualized this study. JL, SF, and KM conducted the literature search. All authors contributed to title and abstract screening. JL, BJ, SF, KM, CM-T, and NK screened the full-text papers. BJ, JL, SF, KM, and PCG extracted the data. BJ, RL, YMW, SF, KM, PCG, CM-T, and NK performed the quality assessment. JL and BJ conducted the formal analysis and visualization. JL, BJ, GCKC, SF, KM, and PCG prepared the original draft. All authors contributed to subsequent drafts, review and editing, and interpretation of the findings. JL, GCKC, BJ, and TS provided supervision. All authors approved the final version and agree to be accountable for all aspects of the work.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CBT</term><def><p>cognitive behavioral therapy</p></def></def-item><def-item><term id="abb2"><italic>DSM</italic></term><def><p><italic>Diagnostic and Statistical Manual of Mental Disorders</italic></p></def></def-item><def-item><term id="abb3"><italic>DSM-5</italic></term><def><p><italic>Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition</italic></p></def></def-item><def-item><term id="abb4"><italic>DSM-5-TR</italic></term><def><p><italic>Diagnostic and Statistical Manual of Mental Disorders (Fifth Edition, Text Revision)</italic></p></def></def-item><def-item><term id="abb5">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb6">GPT</term><def><p>generative pretrained transformer</p></def></def-item><def-item><term id="abb7">GRADE</term><def><p>Grading of Recommendations, Assessment, Development, and Evaluation</p></def></def-item><def-item><term id="abb8"><italic>ICD</italic></term><def><p><italic>International Classification of Diseases</italic></p></def></def-item><def-item><term id="abb9"><italic>ICD-11</italic></term><def><p><italic>International Classification of Diseases, 11th Revision</italic></p></def></def-item><def-item><term id="abb10">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb11">MMAT</term><def><p>Mixed Methods Appraisal Tool</p></def></def-item><def-item><term id="abb12">OCD</term><def><p>obsessive-compulsive disorder</p></def></def-item><def-item><term id="abb13">PI</term><def><p>prediction interval</p></def></def-item><def-item><term id="abb14">PRISMA </term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb15">PRISMA-S</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses literature search extension</p></def></def-item><def-item><term id="abb16">PROSPERO </term><def><p>International Prospective Register of Systematic Reviews</p></def></def-item><def-item><term id="abb17">PTSD </term><def><p>posttraumatic stress disorder</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>GBD 2019 Mental Disorders Collaborators</collab></person-group><article-title>Global, regional, and national burden of 12 mental disorders in 204 countries and territories, 1990&#x2013;2019: a systematic analysis for the Global Burden of Disease Study 2019</article-title><source>Lancet Psychiatry</source><year>2022</year><month>02</month><volume>9</volume><issue>2</issue><fpage>137</fpage><lpage>150</lpage><pub-id pub-id-type="doi">10.1016/S2215-0366(21)00395-3</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kakuma</surname><given-names>R</given-names> </name><name name-style="western"><surname>Minas</surname><given-names>H</given-names> </name><name name-style="western"><surname>van Ginneken</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Human resources for mental health care: current situation and strategies for action</article-title><source>Lancet</source><year>2011</year><month>11</month><day>5</day><volume>378</volume><issue>9803</issue><fpage>1654</fpage><lpage>1663</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(11)61093-3</pub-id><pub-id pub-id-type="medline">22008420</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rameez</surname><given-names>S</given-names> </name><name name-style="western"><surname>Nasir</surname><given-names>A</given-names> </name></person-group><article-title>Barriers to mental health treatment in primary care practice in low- and middle-income countries in a post-COVID era: a systematic review</article-title><source>J Family Med Prim Care</source><year>2023</year><month>08</month><volume>12</volume><issue>8</issue><fpage>1485</fpage><lpage>1504</lpage><pub-id pub-id-type="doi">10.4103/jfmpc.jfmpc_391_22</pub-id><pub-id pub-id-type="medline">37767443</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hoffmann</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Attridge</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Carroll</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Simon</surname><given-names>NJE</given-names> </name><name name-style="western"><surname>Beck</surname><given-names>AF</given-names> </name><name name-style="western"><surname>Alpern</surname><given-names>ER</given-names> </name></person-group><article-title>Association of youth suicides and county-level mental health professional shortage areas in the US</article-title><source>JAMA Pediatr</source><year>2023</year><month>01</month><day>1</day><volume>177</volume><issue>1</issue><fpage>71</fpage><lpage>80</lpage><pub-id pub-id-type="doi">10.1001/jamapediatrics.2022.4419</pub-id><pub-id pub-id-type="medline">36409484</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bedi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Orr-Ewing</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Testing and evaluation of health care applications of large language models: a systematic review</article-title><source>JAMA</source><year>2025</year><month>01</month><day>28</day><volume>333</volume><issue>4</issue><fpage>319</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1001/jama.2024.21700</pub-id><pub-id pub-id-type="medline">39405325</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ni</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Applications and concerns of ChatGPT and other conversational large language models in health care: systematic review</article-title><source>J Med Internet Res</source><year>2024</year><volume>26</volume><fpage>e22769</fpage><pub-id pub-id-type="doi">10.2196/22769</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>First</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Tasman</surname><given-names>A</given-names> </name></person-group><source>Clinical Guide to the Diagnosis and Treatment of Mental Disorders</source><year>2010</year><publisher-name>John Wiley &#x0026; Sons</publisher-name><pub-id pub-id-type="other">978-0-470-74520-5</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hoffman</surname><given-names>RE</given-names> </name><name name-style="western"><surname>Stopek</surname><given-names>S</given-names> </name><name name-style="western"><surname>Andreasen</surname><given-names>NC</given-names> </name></person-group><article-title>A comparative study of manic vs schizophrenic speech disorganization</article-title><source>Arch Gen Psychiatry</source><year>1986</year><month>09</month><volume>43</volume><issue>9</issue><fpage>831</fpage><lpage>838</lpage><pub-id pub-id-type="doi">10.1001/archpsyc.1986.01800090017003</pub-id><pub-id pub-id-type="medline">3753163</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hua</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Na</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A scoping review of large language models for generative tasks in mental health care</article-title><source>npj Digital Med</source><year>2025</year><month>04</month><day>30</day><volume>8</volume><issue>1</issue><fpage>230</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01611-4</pub-id><pub-id pub-id-type="medline">40307331</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Bhanushali</surname><given-names>T</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Badami</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hightow-Weidman</surname><given-names>L</given-names> </name></person-group><article-title>Evaluating generative AI in mental health: systematic review of capabilities and limitations</article-title><source>JMIR Ment Health</source><year>2025</year><month>05</month><day>15</day><volume>12</volume><issue>1</issue><fpage>e70014</fpage><pub-id pub-id-type="doi">10.2196/70014</pub-id><pub-id pub-id-type="medline">40373033</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Thygesen</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Farrington</surname><given-names>J</given-names> </name><name name-style="western"><surname>Keen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name></person-group><article-title>Large language models for mental health applications: systematic review</article-title><source>JMIR Ment Health</source><year>2024</year><volume>11</volume><issue>1</issue><fpage>e57400</fpage><pub-id pub-id-type="doi">10.2196/57400</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>G</given-names> </name></person-group><article-title>The application and ethical implication of generative AI in mental health: systematic review</article-title><source>JMIR Ment Health</source><year>2025</year><volume>12</volume><fpage>e70610</fpage><pub-id pub-id-type="doi">10.2196/70610</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xian</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Xiang</surname><given-names>YT</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>MT</given-names> </name></person-group><article-title>Debate and dilemmas regarding generative AI in mental health care: scoping review</article-title><source>Interact J Med Res</source><year>2024</year><month>08</month><day>12</day><volume>13</volume><issue>1</issue><fpage>e53672</fpage><pub-id pub-id-type="doi">10.2196/53672</pub-id><pub-id pub-id-type="medline">39133916</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kolding</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lundin</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Hansen</surname><given-names>L</given-names> </name><name name-style="western"><surname>&#x00D8;stergaard</surname><given-names>SD</given-names> </name></person-group><article-title>Use of generative artificial intelligence (AI) in psychiatry and mental health care: a systematic review</article-title><source>Acta Neuropsychiatr</source><year>2025</year><volume>37</volume><fpage>e37</fpage><pub-id pub-id-type="doi">10.1017/neu.2024.50</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>P</given-names> </name><etal/></person-group><article-title>The applications of large language models in mental health: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>5</day><volume>27</volume><issue>1</issue><fpage>e69284</fpage><pub-id pub-id-type="doi">10.2196/69284</pub-id><pub-id pub-id-type="medline">40324177</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chung</surname><given-names>NC</given-names> </name><name name-style="western"><surname>Dyer</surname><given-names>G</given-names> </name><name name-style="western"><surname>Brocki</surname><given-names>L</given-names> </name></person-group><article-title>Challenges of large language models for mental health counseling</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 23, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2311.13857</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><name name-style="western"><surname>Song</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Memory in large language models: mechanisms, evaluation and evolution</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 23, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2509.18868</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="other"><person-group person-group-type="author"><collab>OpenAI</collab></person-group><article-title>OpenAI o1 system card</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 30, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.16720</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Alyakin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Seas</surname><given-names>A</given-names> </name><etal/></person-group><article-title>LLM-assisted systematic review of large language models in clinical medicine</article-title><source>Nat Med</source><year>2026</year><month>03</month><volume>32</volume><issue>3</issue><fpage>1152</fpage><lpage>1159</lpage><pub-id pub-id-type="doi">10.1038/s41591-026-04229-5</pub-id><pub-id pub-id-type="medline">41776077</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Page</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>McKenzie</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Bossuyt</surname><given-names>PM</given-names> </name><etal/></person-group><article-title>The PRISMA 2020 statement: an updated guideline for reporting systematic reviews</article-title><source>BMJ</source><year>2021</year><month>03</month><day>29</day><volume>372</volume><fpage>n71</fpage><pub-id pub-id-type="doi">10.1136/bmj.n71</pub-id><pub-id pub-id-type="medline">33782057</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chan</surname><given-names>GCK</given-names> </name><name name-style="western"><surname>He</surname><given-names>E</given-names> </name><name name-style="western"><surname>Leung</surname><given-names>J</given-names> </name><name name-style="western"><surname>Verspoor</surname><given-names>K</given-names> </name></person-group><article-title>A comprehensive systematic review dataset is a rich resource for training and evaluation of AI systems for title and abstract screening</article-title><source>Res Synth Methods</source><year>2025</year><month>03</month><volume>16</volume><issue>2</issue><fpage>308</fpage><lpage>322</lpage><pub-id pub-id-type="doi">10.1017/rsm.2025.1</pub-id><pub-id pub-id-type="medline">41626972</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leung</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>T</given-names> </name><name name-style="western"><surname>Stjepanovic</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Generative artificial intelligence with youth codesign to create vaping awareness advertisements</article-title><source>JAMA Netw Open</source><year>2025</year><month>07</month><day>1</day><volume>8</volume><issue>7</issue><fpage>e2514040</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.14040</pub-id><pub-id pub-id-type="medline">40742592</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Erinoso</surname><given-names>O</given-names> </name></person-group><article-title>Generative artificial intelligence for tobacco health promotion</article-title><source>JAMA Netw Open</source><year>2025</year><month>07</month><day>1</day><volume>8</volume><issue>7</issue><fpage>e2514047</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.14047</pub-id><pub-id pub-id-type="medline">40742599</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hong</surname><given-names>QN</given-names> </name><name name-style="western"><surname>F&#x00E0;bregues</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bartlett</surname><given-names>G</given-names> </name><etal/></person-group><article-title>The Mixed Methods Appraisal Tool (MMAT) version 2018 for information professionals and researchers</article-title><source>EFI</source><year>2018</year><volume>34</volume><issue>4</issue><fpage>285</fpage><lpage>291</lpage><pub-id pub-id-type="doi">10.3233/EFI-180221</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Diaz</surname><given-names>M</given-names> </name></person-group><article-title>Performance measures of the bivariate random effects model for meta-analyses of diagnostic accuracy</article-title><source>Comput Stat Data Anal</source><year>2015</year><month>03</month><volume>83</volume><fpage>82</fpage><lpage>90</lpage><pub-id pub-id-type="doi">10.1016/j.csda.2014.09.021</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sterne</surname><given-names>JAC</given-names> </name><name name-style="western"><surname>Sutton</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ioannidis</surname><given-names>JPA</given-names> </name><etal/></person-group><article-title>Recommendations for examining and interpreting funnel plot asymmetry in meta-analyses of randomised controlled trials</article-title><source>BMJ</source><year>2011</year><month>07</month><day>22</day><volume>343</volume><fpage>d4002</fpage><pub-id pub-id-type="doi">10.1136/bmj.d4002</pub-id><pub-id pub-id-type="medline">21784880</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>IntHout</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ioannidis</surname><given-names>JPA</given-names> </name><name name-style="western"><surname>Borm</surname><given-names>GF</given-names> </name><name name-style="western"><surname>Goeman</surname><given-names>JJ</given-names> </name></person-group><article-title>Small studies are more heterogeneous than large ones: a meta-meta-analysis</article-title><source>J Clin Epidemiol</source><year>2015</year><month>08</month><volume>68</volume><issue>8</issue><fpage>860</fpage><lpage>869</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2015.03.017</pub-id><pub-id pub-id-type="medline">25959635</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Borenstein</surname><given-names>M</given-names> </name><name name-style="western"><surname>Higgins</surname><given-names>JPT</given-names> </name><name name-style="western"><surname>Hedges</surname><given-names>LV</given-names> </name><name name-style="western"><surname>Rothstein</surname><given-names>HR</given-names> </name></person-group><article-title>Basics of meta-analysis: I<sup>2</sup> is not an absolute measure of heterogeneity</article-title><source>Res Synth Methods</source><year>2017</year><month>03</month><volume>8</volume><issue>1</issue><fpage>5</fpage><lpage>18</lpage><pub-id pub-id-type="doi">10.1002/jrsm.1230</pub-id><pub-id pub-id-type="medline">28058794</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Sch&#x00FC;nemann</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Higgins</surname><given-names>JPT</given-names> </name><name name-style="western"><surname>Vist</surname><given-names>GE</given-names> </name><etal/></person-group><article-title>Completing &#x2018;summary of findings&#x2019; tables and grading the certainty of the evidence</article-title><source>Cochrane Handbook for Systematic Reviews of Interventions</source><year>2019</year><publisher-name>Cochrane</publisher-name><fpage>375</fpage><lpage>402</lpage><pub-id pub-id-type="doi">10.1002/9781119536604.ch14</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bartal</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jagodnik</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Dekel</surname><given-names>S</given-names> </name></person-group><article-title>AI and narrative embeddings detect PTSD following childbirth via birth stories</article-title><source>Sci Rep</source><year>2024</year><month>04</month><day>11</day><volume>14</volume><issue>1</issue><fpage>8336</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-54242-2</pub-id><pub-id pub-id-type="medline">38605073</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Danner</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hadzic</surname><given-names>B</given-names> </name><name name-style="western"><surname>Gerhardt</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Advancing mental health diagnostics: GPT-based method for depression detection</article-title><conf-name>2023 62nd Annual Conference of the Society of Instrument and Control Engineers (SICE)</conf-name><conf-date>Sep 6-9, 2023</conf-date><pub-id pub-id-type="doi">10.23919/SICE59929.2023.10354236</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ruan</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Exploiting ChatGPT for diagnosing autism-associated language disorders and identifying distinct features</article-title><source>Research Square</source><comment>Preprint posted online on  May 20, 2024</comment><pub-id pub-id-type="doi">10.21203/rs.3.rs-4359726/v1</pub-id><pub-id pub-id-type="medline">38826194</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lorenzoni</surname><given-names>G</given-names> </name><name name-style="western"><surname>Velmovitsky</surname><given-names>PE</given-names> </name><name name-style="western"><surname>Alencar</surname><given-names>P</given-names> </name><name name-style="western"><surname>Cowan</surname><given-names>D</given-names> </name></person-group><article-title>GPT-4 on clinic depression assessment: an LLM-based pilot study</article-title><conf-name>2024 IEEE International Conference on Big Data (BigData)</conf-name><conf-date>Dec 15-18, 2024</conf-date><conf-loc>Washington, DC</conf-loc><fpage>5043</fpage><lpage>5049</lpage><pub-id pub-id-type="doi">10.1109/BigData62323.2024.10825184</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kaliosis</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ganesan</surname><given-names>AV</given-names> </name><name name-style="western"><surname>Kjell</surname><given-names>ONE</given-names> </name><etal/></person-group><article-title>A systematic evaluation of large language models for PTSD severity estimation: the role of contextual knowledge and modeling strategies</article-title><source>Research Square</source><comment>Preprint posted online on  Dec 24, 2025</comment><pub-id pub-id-type="doi">10.21203/rs.3.rs-8376581/v1</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Teferra</surname><given-names>BG</given-names> </name><name name-style="western"><surname>Perivolaris</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hsiang</surname><given-names>WN</given-names> </name><etal/></person-group><article-title>Leveraging large language models for automated depression screening</article-title><source>PLOS Digital Health</source><year>2025</year><month>07</month><volume>4</volume><issue>7</issue><fpage>e0000943</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000943</pub-id><pub-id pub-id-type="medline">40720397</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Han</surname><given-names>J</given-names> </name><name name-style="western"><surname>Woo</surname><given-names>CW</given-names> </name></person-group><article-title>Interpretable depression assessment using a large language model</article-title><source>PLOS Digital Health</source><year>2026</year><month>02</month><volume>5</volume><issue>2</issue><fpage>e0001205</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0001205</pub-id><pub-id pub-id-type="medline">41662257</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jung</surname><given-names>W</given-names> </name></person-group><article-title>Using large language models to detect depression from user-generated diary text data as a novel approach in digital mental health screening: instrument validation study</article-title><source>J Med Internet Res</source><year>2024</year><month>09</month><day>18</day><volume>26</volume><fpage>e54617</fpage><pub-id pub-id-type="doi">10.2196/54617</pub-id><pub-id pub-id-type="medline">39292502</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhagat</surname><given-names>N</given-names> </name><name name-style="western"><surname>Mackey</surname><given-names>O</given-names> </name><name name-style="western"><surname>Wilcox</surname><given-names>A</given-names> </name></person-group><article-title>Large language models for efficient medical information extraction</article-title><source>AMIA Jt Summits Transl Sci Proc</source><year>2024</year><volume>2024</volume><fpage>509</fpage><lpage>514</lpage><pub-id pub-id-type="medline">38827084</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McCoy</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Perlis</surname><given-names>RH</given-names> </name></person-group><article-title>Applying large language models to stratify suicide risk using narrative clinical notes</article-title><source>J Mood Anxiety Disord</source><year>2025</year><month>06</month><volume>10</volume><fpage>100109</fpage><pub-id pub-id-type="doi">10.1016/j.xjmad.2025.100109</pub-id><pub-id pub-id-type="medline">40657592</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McCoy</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Perlis</surname><given-names>RH</given-names> </name></person-group><article-title>Reasoning language models for more transparent prediction of suicide risk</article-title><source>BMJ Ment Health</source><year>2025</year><month>05</month><day>11</day><volume>28</volume><issue>1</issue><fpage>e301654</fpage><pub-id pub-id-type="doi">10.1136/bmjment-2025-301654</pub-id><pub-id pub-id-type="medline">40350181</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lho</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Park</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Large language models and text embeddings for detecting depression and suicide in patient narratives</article-title><source>JAMA Netw Open</source><year>2025</year><month>05</month><day>1</day><volume>8</volume><issue>5</issue><fpage>e2511922</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.11922</pub-id><pub-id pub-id-type="medline">40408109</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Long</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Large language models for psychiatric diagnosis based on multicenter real-world clinical records: comparative study</article-title><source>JMIR Med Inf</source><year>2026</year><month>01</month><day>13</day><volume>14</volume><fpage>e77699</fpage><pub-id pub-id-type="doi">10.2196/77699</pub-id><pub-id pub-id-type="medline">41408781</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marengo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Longobardi</surname><given-names>C</given-names> </name></person-group><article-title>Detecting suicidal ideation in adolescence using self-reported emotional and behavioral patterns: comparing machine learning and large language model predictions</article-title><source>Assessment</source><year>2025</year><month>12</month><day>31</day><volume>0</volume><fpage>10731911251406405</fpage><pub-id pub-id-type="doi">10.1177/10731911251406405</pub-id><pub-id pub-id-type="medline">41472619</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hur</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Heffner</surname><given-names>J</given-names> </name><name name-style="western"><surname>Feng</surname><given-names>GW</given-names> </name><name name-style="western"><surname>Joormann</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rutledge</surname><given-names>RB</given-names> </name></person-group><article-title>Language sentiment predicts changes in depressive symptoms</article-title><source>Proc Natl Acad Sci U S A</source><year>2024</year><month>09</month><day>24</day><volume>121</volume><issue>39</issue><fpage>e2321321121</fpage><pub-id pub-id-type="doi">10.1073/pnas.2321321121</pub-id><pub-id pub-id-type="medline">39284070</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bi</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name></person-group><article-title>Evaluating the efficacy of AI-based interactive assessments using large language models for depression screening: development and usability study</article-title><source>JMIR Form Res</source><year>2026</year><month>01</month><day>13</day><volume>10</volume><fpage>e78401</fpage><pub-id pub-id-type="doi">10.2196/78401</pub-id><pub-id pub-id-type="medline">41529832</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Heinz</surname><given-names>MV</given-names> </name><name name-style="western"><surname>Bhattacharya</surname><given-names>S</given-names> </name><name name-style="western"><surname>Trudeau</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Testing domain knowledge and risk of bias of a large-scale general artificial intelligence model in mental health</article-title><source>Digital Health</source><year>2023</year><volume>9</volume><fpage>20552076231170499</fpage><pub-id pub-id-type="doi">10.1177/20552076231170499</pub-id><pub-id pub-id-type="medline">37101589</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gargari</surname><given-names>OK</given-names> </name><name name-style="western"><surname>Fatehi</surname><given-names>F</given-names> </name><name name-style="western"><surname>Mohammadi</surname><given-names>I</given-names> </name><name name-style="western"><surname>Firouzabadi</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Shafiee</surname><given-names>A</given-names> </name><name name-style="western"><surname>Habibi</surname><given-names>G</given-names> </name></person-group><article-title>Diagnostic accuracy of large language models in psychiatry</article-title><source>Asian J Psychiatr</source><year>2024</year><month>10</month><volume>100</volume><fpage>104168</fpage><pub-id pub-id-type="doi">10.1016/j.ajp.2024.104168</pub-id><pub-id pub-id-type="medline">39111087</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Sarma</surname><given-names>KV</given-names> </name><name name-style="western"><surname>Hanss</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Halls</surname><given-names>AJM</given-names> </name><name name-style="western"><surname>Becker</surname><given-names>DF</given-names> </name><name name-style="western"><surname>Glowinski</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Krystal</surname><given-names>A</given-names> </name></person-group><article-title>Simulated reasoning and self-verification in generalist large language models for psychiatric diagnostic performance: cross-sectional study</article-title><source>medRxiv</source><comment>Preprint posted online on  Sep 9, 2025</comment><pub-id pub-id-type="doi">10.1101/2025.09.05.25335196</pub-id><pub-id pub-id-type="medline">40963745</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Heston</surname><given-names>TF</given-names> </name></person-group><article-title>Safety of large language models in addressing depression</article-title><source>Cureus</source><year>2023</year><volume>15</volume><issue>12</issue><fpage>e50729</fpage><pub-id pub-id-type="doi">10.7759/cureus.50729</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name><name name-style="western"><surname>Elyoseph</surname><given-names>Z</given-names> </name></person-group><article-title>Suicide risk assessments through the eyes of ChatGPT-3.5 versus ChatGPT-4: vignette study</article-title><source>JMIR Ment Health</source><year>2023</year><month>09</month><day>20</day><volume>10</volume><fpage>e51232</fpage><pub-id pub-id-type="doi">10.2196/51232</pub-id><pub-id pub-id-type="medline">37728984</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lauderdale</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Schmitt</surname><given-names>R</given-names> </name><name name-style="western"><surname>Wuckovich</surname><given-names>B</given-names> </name><name name-style="western"><surname>Dalal</surname><given-names>N</given-names> </name><name name-style="western"><surname>Desai</surname><given-names>H</given-names> </name><name name-style="western"><surname>Tomlinson</surname><given-names>S</given-names> </name></person-group><article-title>Effectiveness of generative AI-large language models&#x2019; recognition of veteran suicide risk: a comparison with human mental health providers using a risk stratification model</article-title><source>Front Psychiatry</source><year>2025</year><volume>16</volume><fpage>1544951</fpage><pub-id pub-id-type="doi">10.3389/fpsyt.2025.1544951</pub-id><pub-id pub-id-type="medline">40248601</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zeng</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>K</given-names> </name></person-group><article-title>Adversarial evaluation algorithm for detecting extreme behaviors of LLMs in psychological counseling scenarios</article-title><conf-name>2025 2nd International Conference on Algorithms, Software Engineering and Network Security (ASENS)</conf-name><conf-date>Mar 21-23, 2025</conf-date><conf-loc>Guangzhou, China</conf-loc><fpage>412</fpage><lpage>415</lpage><pub-id pub-id-type="doi">10.1109/ASENS64990.2025.11011189</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name><name name-style="western"><surname>Elyoseph</surname><given-names>Z</given-names> </name></person-group><article-title>Identifying depression and its determinants upon initiating treatment: ChatGPT versus primary care physicians</article-title><source>Fam Med Community Health</source><year>2023</year><month>09</month><volume>11</volume><issue>4</issue><fpage>e002391</fpage><pub-id pub-id-type="doi">10.1136/fmch-2023-002391</pub-id><pub-id pub-id-type="medline">37844967</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name><name name-style="western"><surname>Rabin</surname><given-names>E</given-names> </name><name name-style="western"><surname>Brann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Elyoseph</surname><given-names>Z</given-names> </name></person-group><article-title>Large language models outperform general practitioners in identifying complex cases of childhood anxiety</article-title><source>Digital Health</source><year>2024</year><volume>10</volume><fpage>20552076241294182</fpage><pub-id pub-id-type="doi">10.1177/20552076241294182</pub-id><pub-id pub-id-type="medline">39687523</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Leonte</surname><given-names>KG</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>ML</given-names> </name><etal/></person-group><article-title>Large language models outperform mental and medical health care professionals in identifying obsessive-compulsive disorder</article-title><source>npj Digital Med</source><year>2024</year><month>07</month><day>19</day><volume>7</volume><issue>1</issue><fpage>193</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01181-x</pub-id><pub-id pub-id-type="medline">39030292</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name></person-group><article-title>Harnessing large language models for identification and treatment of obsessive-compulsive disorder</article-title><source>Comput Hum Behav: Artif Hum</source><year>2025</year><month>12</month><volume>6</volume><fpage>100212</fpage><pub-id pub-id-type="doi">10.1016/j.chbah.2025.100212</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name></person-group><article-title>Evaluating diagnostic accuracy and treatment efficacy in mental health: a comparative analysis of large language model tools and mental health professionals</article-title><source>Eur J Invest Health Psychol Educ</source><year>2025</year><month>01</month><day>18</day><volume>15</volume><issue>1</issue><fpage>9</fpage><pub-id pub-id-type="doi">10.3390/ejihpe15010009</pub-id><pub-id pub-id-type="medline">39852192</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name><name name-style="western"><surname>Haber</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Levi-Belz</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Elyoseph</surname><given-names>Z</given-names> </name></person-group><article-title>A step toward the future? Evaluating GenAI QPR simulation training for mental health gatekeepers</article-title><source>Front Med (Lausanne)</source><year>2025</year><volume>12</volume><fpage>1599900</fpage><pub-id pub-id-type="doi">10.3389/fmed.2025.1599900</pub-id><pub-id pub-id-type="medline">40568211</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>D&#x2019;Souza</surname><given-names>RF</given-names> </name><name name-style="western"><surname>Amanullah</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mathew</surname><given-names>M</given-names> </name><name name-style="western"><surname>Surapaneni</surname><given-names>KM</given-names> </name></person-group><article-title>Appraising the performance of ChatGPT in psychiatry using 100 clinical case vignettes</article-title><source>Asian J Psychiatr</source><year>2023</year><month>11</month><volume>89</volume><fpage>103770</fpage><pub-id pub-id-type="doi">10.1016/j.ajp.2023.103770</pub-id><pub-id pub-id-type="medline">37812998</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>KW</given-names> </name><name name-style="western"><surname>Rostam-Abadi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chaudhary</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Evaluating diagnostic accuracy and clinical reasoning of multiple large language models in psychiatry</article-title><source>medRxiv</source><comment>Preprint posted online on  Feb 11, 2026</comment><pub-id pub-id-type="doi">10.64898/2026.02.03.26345402</pub-id><pub-id pub-id-type="medline">41728283</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sarma</surname><given-names>KV</given-names> </name><name name-style="western"><surname>Hanss</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Halls</surname><given-names>AJM</given-names> </name><etal/></person-group><article-title>Integrating expert knowledge into large language models improves performance for psychiatric reasoning and diagnosis</article-title><source>Psychiatry Res</source><year>2026</year><month>01</month><volume>355</volume><fpage>116844</fpage><pub-id pub-id-type="doi">10.1016/j.psychres.2025.116844</pub-id><pub-id pub-id-type="medline">41270691</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>CBT-bench: evaluating large language models on assisting cognitive behavior therapy</article-title><conf-name>Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics</conf-name><conf-date>Apr 29 to May 4, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.naacl-long.196</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Aleem</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zahoor</surname><given-names>I</given-names> </name><name name-style="western"><surname>Naseem</surname><given-names>M</given-names> </name></person-group><article-title>Towards culturally adaptive large language models in mental health: using ChatGPT as a case study</article-title><conf-name>CSCW Companion &#x2019;24: Companion Publication of the 2024 Conference on Computer-Supported Cooperative Work and Social Computing</conf-name><conf-date>Nov 9-13, 2024</conf-date><conf-loc>San Jose Costa Rica</conf-loc><fpage>240</fpage><lpage>247</lpage><pub-id pub-id-type="doi">10.1145/3678884.3681858</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>WYC</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>G</given-names> </name></person-group><article-title>Understanding medical information and emotional support needs in mental health questions with large language models</article-title><source>Ind Manage Data Syst</source><year>2026</year><month>06</month><day>22</day><volume>126</volume><issue>7</issue><fpage>2205</fpage><lpage>2230</lpage><pub-id pub-id-type="doi">10.1108/IMDS-05-2025-0609</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elyoseph</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name></person-group><article-title>Comparing the perspectives of generative AI, mental health experts, and the general public on schizophrenia recovery: case vignette study</article-title><source>JMIR Ment Health</source><year>2024</year><volume>11</volume><fpage>e53043</fpage><lpage>e53043</lpage><pub-id pub-id-type="doi">10.2196/53043</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Perlis</surname><given-names>RH</given-names> </name></person-group><article-title>Research letter: application of GPT-4 to select next-step antidepressant treatment in major depression</article-title><source>medRxiv</source><comment>Preprint posted online on  Apr 18, 2023</comment><pub-id pub-id-type="doi">10.1101/2023.04.14.23288595</pub-id><pub-id pub-id-type="medline">37131648</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Perlis</surname><given-names>RH</given-names> </name><name name-style="western"><surname>Goldberg</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Ostacher</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Schneck</surname><given-names>CD</given-names> </name></person-group><article-title>Clinical decision support for bipolar depression using large language models</article-title><source>Neuropsychopharmacology</source><year>2024</year><month>08</month><volume>49</volume><issue>9</issue><fpage>1412</fpage><lpage>1416</lpage><pub-id pub-id-type="doi">10.1038/s41386-024-01841-2</pub-id><pub-id pub-id-type="medline">38480911</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Silva</surname><given-names>R</given-names> </name><name name-style="western"><surname>Gomes</surname><given-names>L</given-names> </name></person-group><article-title>An adaptive language model-based intelligent medication assistant for the decision support of antidepressant prescriptions</article-title><source>Comput Biol Med</source><year>2025</year><month>05</month><volume>190</volume><fpage>110065</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2025.110065</pub-id><pub-id pub-id-type="medline">40147190</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McBain</surname><given-names>RK</given-names> </name><name name-style="western"><surname>Cantor</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>LA</given-names> </name><etal/></person-group><article-title>Evaluation of alignment between large language models and expert clinicians in suicide risk assessment</article-title><source>Psychiatr Serv</source><year>2025</year><month>11</month><day>1</day><volume>76</volume><issue>11</issue><fpage>944</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1176/appi.ps.20250086</pub-id><pub-id pub-id-type="medline">41174947</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tan</surname><given-names>KS</given-names> </name><name name-style="western"><surname>Cervin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Leman</surname><given-names>P</given-names> </name><name name-style="western"><surname>Nielsen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>PV</given-names> </name><name name-style="western"><surname>Medvedev</surname><given-names>O</given-names> </name></person-group><article-title>AI meets psychology: an exploratory study of large language models&#x2019; competence in psychotherapy contexts</article-title><source>J Psychol AI</source><year>2025</year><month>12</month><day>31</day><volume>1</volume><issue>1</issue><fpage>2545258</fpage><pub-id pub-id-type="doi">10.1080/29974100.2025.2545258</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thotapalli</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yilanli</surname><given-names>M</given-names> </name><name name-style="western"><surname>McKay</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Potential of ChatGPT in youth mental health emergency triage: comparative analysis with clinicians</article-title><source>PCN Rep</source><year>2025</year><month>09</month><volume>4</volume><issue>3</issue><fpage>e70159</fpage><pub-id pub-id-type="doi">10.1002/pcn5.70159</pub-id><pub-id pub-id-type="medline">40673126</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Acevedo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Aneja</surname><given-names>E</given-names> </name><name name-style="western"><surname>Opler</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Valera</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jarmon</surname><given-names>E</given-names> </name></person-group><article-title>Evaluating the efficacy of ChatGPT-3.5 versus human-delivered text-based cognitive-behavioral therapy: a comparative pilot study</article-title><source>Am J Psychother</source><year>2026</year><month>03</month><day>1</day><volume>79</volume><issue>1</issue><fpage>4</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.1176/appi.psychotherapy.20240070</pub-id><pub-id pub-id-type="medline">40820507</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alanzi</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Alharthi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Alrumman</surname><given-names>S</given-names> </name><etal/></person-group><article-title>ChatGPT as a psychotherapist for anxiety disorders: an empirical study with anxiety patients</article-title><source>Nutr Health</source><year>2025</year><month>09</month><volume>31</volume><issue>3</issue><fpage>1111</fpage><lpage>1123</lpage><pub-id pub-id-type="doi">10.1177/02601060241281906</pub-id><pub-id pub-id-type="medline">39370914</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haber</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Levi-Belz</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Elbak</surname><given-names>YS</given-names> </name><name name-style="western"><surname>Elyoseph</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name></person-group><article-title>Validating GenAI feedback in suicide prevention training: a mixed-methods study of QPR skill assessment</article-title><source>Front Med (Lausanne)</source><year>2025</year><volume>12</volume><fpage>1709743</fpage><pub-id pub-id-type="doi">10.3389/fmed.2025.1709743</pub-id><pub-id pub-id-type="medline">41625774</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hodson</surname><given-names>N</given-names> </name><name name-style="western"><surname>Williamson</surname><given-names>S</given-names> </name></person-group><article-title>Can large language models replace therapists? Evaluating performance at simple cognitive behavioral therapy tasks</article-title><source>JMIR AI</source><year>2024</year><month>07</month><day>30</day><volume>3</volume><fpage>e52500</fpage><pub-id pub-id-type="doi">10.2196/52500</pub-id><pub-id pub-id-type="medline">39078696</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jain</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sandhu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>G</given-names> </name><name name-style="western"><surname>Rakhra</surname><given-names>M</given-names> </name></person-group><article-title>The role of AI counselling in journaling for mental health improvement</article-title><conf-name>2024 International Conference on Electrical Electronics and Computing Technologies (ICEECT)</conf-name><conf-date>Aug 29-31, 2024</conf-date><conf-loc>Greater Noida, India</conf-loc><fpage>1</fpage><lpage>6</lpage><pub-id pub-id-type="doi">10.1109/ICEECT61758.2024.10739128</pub-id></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Napiwotzki</surname><given-names>I</given-names> </name><name name-style="western"><surname>Laue</surname><given-names>J</given-names> </name><name name-style="western"><surname>Caldarone</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Comparing human and AI therapists in behavioral activation for depression: cross-sectional questionnaire study</article-title><source>JMIR Form Res</source><year>2025</year><month>12</month><day>4</day><volume>9</volume><fpage>e78138</fpage><pub-id pub-id-type="doi">10.2196/78138</pub-id><pub-id pub-id-type="medline">41343763</pub-id></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Moell</surname><given-names>B</given-names> </name></person-group><article-title>Comparing the efficacy of GPT-4 and Chat-GPT in mental health care: a blind assessment of large language models for psychological support (preprint)</article-title><source>JMIR Ment Health</source><comment>Preprint posted online on  Mar 20, 2023</comment><pub-id pub-id-type="doi">10.2196/preprints.47439</pub-id><pub-id pub-id-type="medline">37114093</pub-id></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Shen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Are large language models possible to conduct cognitive behavioral therapy?</article-title><conf-name>2024 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name><conf-date>Dec 3-6, 2024</conf-date><conf-loc>Lisbon, Portugal</conf-loc><fpage>3695</fpage><lpage>3700</lpage><pub-id pub-id-type="doi">10.1109/BIBM62325.2024.10821773</pub-id></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Adhikary</surname><given-names>PK</given-names> </name><name name-style="western"><surname>Srivastava</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Exploring the efficacy of large language models in summarizing mental health counseling sessions: benchmark study</article-title><source>JMIR Ment Health</source><year>2024</year><month>07</month><day>23</day><volume>11</volume><fpage>e57306</fpage><pub-id pub-id-type="doi">10.2196/57306</pub-id><pub-id pub-id-type="medline">39042893</pub-id></nlm-citation></ref><ref id="ref81"><label>81</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cardamone</surname><given-names>NC</given-names> </name><name name-style="western"><surname>Olfson</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schmutte</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Classifying unstructured text in electronic health records for mental health prediction models: large language model evaluation study</article-title><source>JMIR Med Inf</source><year>2025</year><month>01</month><day>21</day><volume>13</volume><fpage>e65454</fpage><pub-id pub-id-type="doi">10.2196/65454</pub-id><pub-id pub-id-type="medline">39864953</pub-id></nlm-citation></ref><ref id="ref82"><label>82</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Donnelly</surname><given-names>HK</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>GK</given-names> </name><name name-style="western"><surname>Green</surname><given-names>KL</given-names> </name><etal/></person-group><article-title>Exploring the potential of large language models for automated safety plan scoring in outpatient mental health settings (preprint)</article-title><source>JMIR Ment Health</source><comment>Preprint posted online on  Sep 3, 2025</comment><pub-id pub-id-type="doi">10.2196/preprints.79010</pub-id><pub-id pub-id-type="medline">40196270</pub-id></nlm-citation></ref><ref id="ref83"><label>83</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gireesh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Shukla</surname><given-names>L</given-names> </name><name name-style="western"><surname>Shivaprakash</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mukherjee</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chand</surname><given-names>P</given-names> </name><name name-style="western"><surname>Murthy</surname><given-names>P</given-names> </name></person-group><article-title>Language models for standardising clinical notes and information extraction in addiction psychiatry-an empirical study</article-title><source>Drug Alcohol Rev</source><year>2026</year><month>01</month><volume>45</volume><issue>1</issue><fpage>e70059</fpage><pub-id pub-id-type="doi">10.1111/dar.70059</pub-id><pub-id pub-id-type="medline">41158037</pub-id></nlm-citation></ref><ref id="ref84"><label>84</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Matsumura</surname><given-names>M</given-names> </name><name name-style="western"><surname>Nishida</surname><given-names>K</given-names> </name><name name-style="western"><surname>Toyoda</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Quantifying improvement of psychotic symptoms in clozapine-treated schizophrenia: clinical note analysis with large language models</article-title><source>Sci Rep</source><year>2026</year><month>02</month><day>13</day><volume>16</volume><issue>1</issue><fpage>8835</fpage><pub-id pub-id-type="doi">10.1038/s41598-026-39676-0</pub-id><pub-id pub-id-type="medline">41680410</pub-id></nlm-citation></ref><ref id="ref85"><label>85</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Janota</surname><given-names>B</given-names> </name><name name-style="western"><surname>Janota</surname><given-names>K</given-names> </name></person-group><article-title>Application of artificial intelligence (AI) in the creation of discharge summaries in psychiatric clinics</article-title><source>Int J Psychiatry Med</source><year>2025</year><month>05</month><volume>60</volume><issue>3</issue><fpage>330</fpage><lpage>337</lpage><pub-id pub-id-type="doi">10.1177/00912174241284730</pub-id><pub-id pub-id-type="medline">39285727</pub-id></nlm-citation></ref><ref id="ref86"><label>86</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>Q</given-names> </name></person-group><article-title>Physician versus large language model chatbot responses to web-based questions from autistic patients in Chinese: cross-sectional comparative analysis</article-title><source>J Med Internet Res</source><year>2024</year><month>04</month><day>30</day><volume>26</volume><fpage>e54706</fpage><pub-id pub-id-type="doi">10.2196/54706</pub-id><pub-id pub-id-type="medline">38687566</pub-id></nlm-citation></ref><ref id="ref87"><label>87</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Development and preliminary evaluation of a virtual standardized patient system for psychiatric interview training</article-title><source>BMC Psychiatry</source><year>2026</year><volume>26</volume><issue>1</issue><fpage>264</fpage><pub-id pub-id-type="doi">10.1186/s12888-026-07896-3</pub-id></nlm-citation></ref><ref id="ref88"><label>88</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yilanli</surname><given-names>M</given-names> </name><name name-style="western"><surname>McKay</surname><given-names>I</given-names> </name><name name-style="western"><surname>Jackson</surname><given-names>DI</given-names> </name><name name-style="western"><surname>Sezgin</surname><given-names>E</given-names> </name></person-group><article-title>Large language models for individualized psychoeducational tools for psychosis: a cross-sectional study</article-title><source>medRxiv</source><comment>Preprint posted online on  Jul 29, 2024</comment><pub-id pub-id-type="doi">10.1101/2024.07.26.24311075</pub-id></nlm-citation></ref><ref id="ref89"><label>89</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Senturk</surname><given-names>E</given-names> </name><name name-style="western"><surname>Koparal</surname><given-names>B</given-names> </name></person-group><article-title>ChatGPT-4o vs psychiatrists in responding to common antidepressant concerns</article-title><source>Am J Health Promot</source><year>2026</year><month>01</month><volume>40</volume><issue>1</issue><fpage>10</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1177/08901171251348208</pub-id><pub-id pub-id-type="medline">40440446</pub-id></nlm-citation></ref><ref id="ref90"><label>90</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barabas</surname><given-names>L</given-names> </name><name name-style="western"><surname>Novotny</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jung</surname><given-names>D</given-names> </name><name name-style="western"><surname>M&#x00FC;ller</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mertse</surname><given-names>NN</given-names> </name></person-group><article-title>Exploring the potential of ChatGPT as a digital advisor in acute psychiatric crises: a feasibility study</article-title><source>Nervenarzt</source><year>2026</year><month>05</month><volume>97</volume><issue>3</issue><fpage>265</fpage><lpage>271</lpage><pub-id pub-id-type="doi">10.1007/s00115-025-01837-3</pub-id><pub-id pub-id-type="medline">40478290</pub-id></nlm-citation></ref><ref id="ref91"><label>91</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McBain</surname><given-names>RK</given-names> </name><name name-style="western"><surname>Cantor</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>LA</given-names> </name><etal/></person-group><article-title>Competency of large language models in evaluating appropriate responses to suicidal ideation: comparative study</article-title><source>J Med Internet Res</source><year>2025</year><month>03</month><day>5</day><volume>27</volume><fpage>e67891</fpage><pub-id pub-id-type="doi">10.2196/67891</pub-id><pub-id pub-id-type="medline">40053817</pub-id></nlm-citation></ref><ref id="ref92"><label>92</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scholich</surname><given-names>T</given-names> </name><name name-style="western"><surname>Barr</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stirman</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Raj</surname><given-names>S</given-names> </name></person-group><article-title>A comparison of responses from human therapists and large language model-based chatbots to assess therapeutic communication: mixed methods study</article-title><source>JMIR Ment Health</source><year>2025</year><month>05</month><day>21</day><volume>12</volume><fpage>e69709</fpage><pub-id pub-id-type="doi">10.2196/69709</pub-id><pub-id pub-id-type="medline">40397927</pub-id></nlm-citation></ref><ref id="ref93"><label>93</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Bannett</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gunturkun</surname><given-names>F</given-names> </name><name name-style="western"><surname>Pillai</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Leveraging a large language model to assess quality-of-care: monitoring ADHD medication side effects</article-title><source>medRxiv</source><comment>Preprint posted online on  Apr 24, 2024</comment><pub-id pub-id-type="doi">10.1101/2024.04.23.24306225</pub-id><pub-id pub-id-type="medline">38712037</pub-id></nlm-citation></ref><ref id="ref94"><label>94</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rao</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Isath</surname><given-names>A</given-names> </name><name name-style="western"><surname>Krishnan</surname><given-names>P</given-names> </name><etal/></person-group><article-title>ChatGPT: a conceptual review of applications and utility in the field of medicine</article-title><source>J Med Syst</source><year>2024</year><month>06</month><day>5</day><volume>48</volume><issue>1</issue><fpage>59</fpage><pub-id pub-id-type="doi">10.1007/s10916-024-02075-x</pub-id><pub-id pub-id-type="medline">38836893</pub-id></nlm-citation></ref><ref id="ref95"><label>95</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Giacobbe</surname><given-names>DR</given-names> </name><name name-style="western"><surname>Marelli</surname><given-names>C</given-names> </name><name name-style="western"><surname>Guastavino</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Artificial intelligence and prescription of antibiotic therapy: present and future</article-title><source>Expert Rev Anti-Infect Ther</source><year>2024</year><month>10</month><volume>22</volume><issue>10</issue><fpage>819</fpage><lpage>833</lpage><pub-id pub-id-type="doi">10.1080/14787210.2024.2386669</pub-id><pub-id pub-id-type="medline">39155449</pub-id></nlm-citation></ref><ref id="ref96"><label>96</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferrag</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Tihanyi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Debbah</surname><given-names>M</given-names> </name></person-group><article-title>From LLM reasoning to autonomous AI agents: a comprehensive review</article-title><source>IEEE Access</source><year>2025</year><volume>14</volume><fpage>84237</fpage><lpage>84285</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2026.3698694</pub-id><pub-id pub-id-type="medline">40292759</pub-id></nlm-citation></ref><ref id="ref97"><label>97</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dergaa</surname><given-names>I</given-names> </name><name name-style="western"><surname>Fekih-Romdhane</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hallit</surname><given-names>S</given-names> </name><etal/></person-group><article-title>ChatGPT is not ready yet for use in providing mental health assessment and interventions</article-title><source>Front Psychiatry</source><year>2023</year><volume>14</volume><fpage>1277756</fpage><pub-id pub-id-type="doi">10.3389/fpsyt.2023.1277756</pub-id><pub-id pub-id-type="medline">38239905</pub-id></nlm-citation></ref><ref id="ref98"><label>98</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elyoseph</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name><name name-style="western"><surname>Shinan-Altman</surname><given-names>S</given-names> </name></person-group><article-title>Assessing prognosis in depression: comparing perspectives of AI models, mental health professionals and the general public</article-title><source>Fam Med Community Health</source><year>2024</year><month>01</month><volume>12</volume><issue>Suppl 1</issue><fpage>e002583</fpage><pub-id pub-id-type="doi">10.1136/fmch-2023-002583</pub-id></nlm-citation></ref><ref id="ref99"><label>99</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Vries</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schaub</surname><given-names>MP</given-names> </name></person-group><article-title>Opportunities and risks of large language models in digital interventions for substance use disorders</article-title><source>Curr Opin Psychiatry</source><year>2026</year><month>07</month><day>1</day><volume>39</volume><issue>4</issue><fpage>308</fpage><lpage>313</lpage><pub-id pub-id-type="doi">10.1097/YCO.0000000000001088</pub-id><pub-id pub-id-type="medline">42083979</pub-id></nlm-citation></ref><ref id="ref100"><label>100</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haltaufderheide</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ranisch</surname><given-names>R</given-names> </name></person-group><article-title>The ethics of ChatGPT in medicine and healthcare: a systematic review on large language models (LLMs)</article-title><source>npj Digital Med</source><year>2024</year><month>07</month><day>8</day><volume>7</volume><issue>1</issue><fpage>183</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01157-x</pub-id><pub-id pub-id-type="medline">38977771</pub-id></nlm-citation></ref><ref id="ref101"><label>101</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dada</surname><given-names>A</given-names> </name><name name-style="western"><surname>Puladi</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kleesiek</surname><given-names>J</given-names> </name><name name-style="western"><surname>Egger</surname><given-names>J</given-names> </name></person-group><article-title>ChatGPT in healthcare: a taxonomy and systematic review</article-title><source>Comput Methods Programs Biomed</source><year>2024</year><month>03</month><volume>245</volume><fpage>108013</fpage><pub-id pub-id-type="doi">10.1016/j.cmpb.2024.108013</pub-id><pub-id pub-id-type="medline">38262126</pub-id></nlm-citation></ref><ref id="ref102"><label>102</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Codreanu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>J</given-names> </name></person-group><article-title>The widespread adoption of large language model-assisted writing across society</article-title><source>Patterns (N Y)</source><year>2025</year><month>12</month><day>12</day><volume>6</volume><issue>12</issue><fpage>101366</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2025.101366</pub-id><pub-id pub-id-type="medline">41472824</pub-id></nlm-citation></ref><ref id="ref103"><label>103</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chan</surname><given-names>GCK</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Causal inference with observational data in addiction research</article-title><source>Addiction</source><year>2022</year><month>10</month><volume>117</volume><issue>10</issue><fpage>2736</fpage><lpage>2744</lpage><pub-id pub-id-type="doi">10.1111/add.15972</pub-id></nlm-citation></ref><ref id="ref104"><label>104</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chan</surname><given-names>GCK</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>T</given-names> </name><name name-style="western"><surname>Stjepanovi&#x0107;</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Designing observational studies for credible causal inference in addiction research-directed acyclic graphs, modified disjunctive cause criterion and target trial emulation</article-title><source>Addiction</source><year>2024</year><month>06</month><volume>119</volume><issue>6</issue><fpage>1125</fpage><lpage>1134</lpage><pub-id pub-id-type="doi">10.1111/add.16442</pub-id><pub-id pub-id-type="medline">38343103</pub-id></nlm-citation></ref><ref id="ref105"><label>105</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Prompt engineering in consistency and reliability with the evidence-based guideline for LLMs</article-title><source>npj Digital Med</source><year>2024</year><month>02</month><day>20</day><volume>7</volume><issue>1</issue><fpage>41</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01029-4</pub-id><pub-id pub-id-type="medline">38378899</pub-id></nlm-citation></ref><ref id="ref106"><label>106</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wright</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pagliaro</surname><given-names>C</given-names> </name><name name-style="western"><surname>Page</surname><given-names>IS</given-names> </name><name name-style="western"><surname>Diminic</surname><given-names>S</given-names> </name></person-group><article-title>A review of excluded groups and non-response in population-based mental health surveys from high-income countries</article-title><source>Soc Psychiatry Psychiatr Epidemiol</source><year>2023</year><month>09</month><volume>58</volume><issue>9</issue><fpage>1265</fpage><lpage>1292</lpage><pub-id pub-id-type="doi">10.1007/s00127-023-02488-y</pub-id><pub-id pub-id-type="medline">37212903</pub-id></nlm-citation></ref><ref id="ref107"><label>107</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chandra</surname><given-names>M</given-names> </name><name name-style="western"><surname>Verma</surname><given-names>G</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>De Choudhury</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>S</given-names> </name></person-group><article-title>Better to ask in English: cross-lingual evaluation of large language models for healthcare queries</article-title><conf-name>WWW &#x2019;24: Proceedings of the ACM Web Conference 2024</conf-name><conf-date>May 13-17, 2024</conf-date><conf-loc>Singapore, Singapore</conf-loc><fpage>2627</fpage><lpage>2638</lpage><pub-id pub-id-type="doi">10.1145/3589334.3645643</pub-id></nlm-citation></ref><ref id="ref108"><label>108</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Nadkarni</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hanlon</surname><given-names>C</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>V</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Tasman</surname><given-names>A</given-names> </name><name name-style="western"><surname>Riba</surname><given-names>MB</given-names></name><name name-style="western"><surname>Alarc&#x00F3;n</surname><given-names>RD</given-names></name><name name-style="western"><surname>Alfonso</surname><given-names>CA</given-names> </name></person-group><article-title>Mental health care models in low-and middle-income countries, in Tasman&#x2019;s Psychiatry</article-title><source>Tasman&#x2019;s Psychiatry</source><year>2023</year><publisher-name>Springer</publisher-name><fpage>1</fpage><lpage>47</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-42825-9_156-1</pub-id></nlm-citation></ref><ref id="ref109"><label>109</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reynolds</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Christie</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Dicks</surname><given-names>LV</given-names> </name><etal/></person-group><article-title>Will AI speed up literature reviews or derail them entirely?</article-title><source>Nature</source><year>2025</year><month>07</month><day>10</day><volume>643</volume><issue>8071</issue><fpage>329</fpage><lpage>331</lpage><pub-id pub-id-type="doi">10.1038/d41586-025-02069-w</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Review methods, meta-analysis plots, and methodological quality assessment.</p><media xlink:href="ai_v5i1e87730_app1.docx" xlink:title="DOCX File, 2353 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Summary of studies on the applications of large language models in mental health care (N=66).</p><media xlink:href="ai_v5i1e87730_app2.docx" xlink:title="DOCX File, 62 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 1</label><p>PRISMA Checklist.</p><media xlink:href="ai_v5i1e87730_app3.docx" xlink:title="DOCX File, 22 KB"/></supplementary-material><supplementary-material id="app4"><label>Checklist 2</label><p>PRISMA-S Checklist.</p><media xlink:href="ai_v5i1e87730_app4.docx" xlink:title="DOCX File, 17 KB"/></supplementary-material></app-group></back></article>