<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e88082</article-id><article-id pub-id-type="doi">10.2196/88082</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Large Language Models for Mental Health Prediction: Scoping Review of Bias and Clinical Utility Documentation in 2019-2024</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Bleuze</surname><given-names>Cl&#x00E9;mentine</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Fort</surname><given-names>Kar&#x00EB;n</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Martin</surname><given-names>Vincent P</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>N&#x00E9;v&#x00E9;ol</surname><given-names>Aur&#x00E9;lie</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Loria, Universit&#x00E9; de Lorraine, CNRS, Inria</institution><addr-line>Loria Campus Scientifique, BP 239</addr-line><addr-line>Nancy</addr-line><country>France</country></aff><aff id="aff2"><institution>LISN, Universit&#x00E9; Paris-Saclay, CNRS</institution><addr-line>Orsay</addr-line><country>France</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Malin</surname><given-names>Bradley</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Cao</surname><given-names>Shihua</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Hu</surname><given-names>Songbo</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Cl&#x00E9;mentine Bleuze, MSc, Loria, Universit&#x00E9; de Lorraine, CNRS, Inria, Loria Campus Scientifique, BP 239, Nancy, F-54000, France, 33 3-83-59-20-20; <email>clementine.bleuze@univ-lorraine.fr</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>13</day><month>8</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e88082</elocation-id><history><date date-type="received"><day>19</day><month>11</month><year>2025</year></date><date date-type="rev-recd"><day>19</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>22</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Cl&#x00E9;mentine Bleuze, Kar&#x00EB;n Fort, Vincent P Martin, Aur&#x00E9;lie N&#x00E9;v&#x00E9;ol. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 13.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e88082"/><abstract><sec><title>Background</title><p>A growing body of literature leverages large language models (LLMs) to make mental health predictions. However, these models are prone to bias, and studies to validate their clinical utility are lacking.</p></sec><sec><title>Objective</title><p>This scoping review aims to uncover bias and clinical utility limitations stemming from the methodological design of LLM-based mental health predictive systems. In addition, it intends to document the level of self-reflection about bias and clinical challenges reported by authors in their own work.</p></sec><sec sec-type="methods"><title>Methods</title><p>This work follows the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews) guidelines and was registered online. Eligible studies were original research articles in English published between 2019 and 2024, using LLMs to detect mental health conditions in nonsynthetic textual data. The search was conducted in 5 scientific databases (PubMed, Web of Science, IEEE Xplore, ACM Digital Library, and ACL Anthology) with queries associating keywords related to &#x201C;Mental Health,&#x201D; &#x201C;Large Language Models,&#x201D; and &#x201C;Prediction.&#x201D; We extracted both methodological information about the included studies and authors&#x2019; statements relevant to issues of bias and clinical utility. This extraction was based on a framework screening the entire pipeline of development of LLMs with applications in mental health: research design and selection, data collection, outcome definition, model development, and postdeployment considerations. Statistical description of the retrieved entities, as well as thematic coding, was performed for analysis.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 2472 articles were identified, of which 263 (10.6%) were assessed for eligibility, and 201 (8.1%) were included in the review. Included studies were mostly recent, indicating a growing interest in the use of LLMs for mental health predictions. Our analysis revealed that a majority of studies share similar methodological choices along their development pipeline: most of them focus on depressive disorders identified via processing user texts on social media, mainly with the use of nonspecialist LLMs derived from BERT (Bidirectional Encoder Representations from Transformers). Following previous works on these matters, we highlighted how these choices may hinder the clinical relevance and fairness of the envisioned systems. Similarly, we found that 164 (81.6%) studies mention themes related to bias and clinical utility; however, most of the discussion revolves around data-centered issues. Only 41 (20.4%) articles mention themes associated with at least 3 out of 5 pipeline steps, suggesting a limited appropriation of the notions of bias and clinical utility in such a sensitive context as mental health analysis.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Bias and clinical utility are lightly covered in the field of LLM-based mental health prediction research as of 2019&#x2010;2024. In-depth approaches involving interdisciplinary teams of clinicians and natural language processing specialists are needed to ensure technical soundness, clinical relevance, and fair outcomes for potential users.</p></sec></abstract><kwd-group><kwd>natural language processing</kwd><kwd>large language models</kwd><kwd>mental health</kwd><kwd>bias</kwd><kwd>clinical utility</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Biomedical natural language processing (NLP) methods aim to support clinical research and practice, with suggested uses spanning information retrieval, medical documentation generation, and patient data analysis [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. As language has been studied as a marker for conditions such as depression [<xref ref-type="bibr" rid="ref3">3</xref>], schizophrenia [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>], and Alzheimer disease [<xref ref-type="bibr" rid="ref6">6</xref>], mental health&#x2013;oriented tasks have become popular in NLP. Notably, methods to analyze patient-authored texts or medical records as indicative of a given condition or symptom have been developed, with the underlying hypothesis that this could translate into clinically useful tools for diagnosis or large-scale screening [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref9">9</xref>]. A number of these methods are now based on large language models (LLMs), which recently garnered global attention from the public, industrials, and researchers for their assumed general language capabilities.</p><p>However, it is increasingly documented that LLMs can produce unreliable results and potentially harm users by perpetuating and amplifying bias [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref14">14</xref>], which has been defined as &#x201C;the presence of systematic errors or disparities within decision-making processes that disproportionately affect specific subgroups&#x201D; [<xref ref-type="bibr" rid="ref11">11</xref>]. While current medical practice is admittedly not exempt from bias itself [<xref ref-type="bibr" rid="ref15">15</xref>], automated systems such as LLMs amplify existing stereotypes, which can lead to unjustified differences in simulated diagnoses and treatment plans for vignettes of patients with different ethnicities and genders [<xref ref-type="bibr" rid="ref13">13</xref>]. Relatedly, it has been reported that male profiles are overgenerated by LLMs (regardless of the true prevalence of the considered condition) both in English and in French [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. Despite substantial research efforts, no reliable mitigation technique has been proved to fully resolve these issues [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>], which may be intrinsic to LLMs [<xref ref-type="bibr" rid="ref18">18</xref>].</p><p>Moreover, existing evaluation settings for biomedical LLMs are limited [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref19">19</xref>] and real-world impact studies are still lacking in the NLP community [<xref ref-type="bibr" rid="ref20">20</xref>], leaving many ethical and regulatory challenges unaddressed [<xref ref-type="bibr" rid="ref21">21</xref>]. As a result, it is still unclear to what extent these LLM-based systems are clinically &#x201C;[useful] in improving patient outcomes, informing clinical decision-making, and optimizing health care resources&#x201D; [<xref ref-type="bibr" rid="ref22">22</xref>] in mental health. In addition, computer scientists and NLP engineers creating such systems may lack the medical expertise needed to make accurate design choices. It seems therefore crucial to methodically audit LLM-based systems both for bias and clinical utility before considering downstream clinical deployments.</p><p>Existing reviews have investigated the applications, benefits, and risks of LLMs in health care [<xref ref-type="bibr" rid="ref23">23</xref>] and mental health [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref26">26</xref>], highlighting general ethical challenges such as data privacy, fairness and bias, clinical integration, and ethical governance. There is, however, a lack of literature that explicitly connects these different issues and systematically assesses them throughout the development steps of the considered systems, taking a distance from indicators of (possibly ill-defined [<xref ref-type="bibr" rid="ref27">27</xref>]) performance. Notably, although they are crucial to understand downstream risks, existing bias studies in health-related scenarios generally posit bias as a single-dimension construct (eg, gender or racial bias) [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref28">28</xref>], using LLMs as readily available tools whose development is left unquestioned. Yet the decision-making processes from which bias can stem are numerous, which calls for a bias investigation throughout the entire development pipeline of the considered system [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. Similarly, the limitations and barriers to the clinical implementation of NLP technologies [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref31">31</xref>] tend to be listed as post hoc challenges of the systems rather than connected with underlying design choices. In this review, we consider bias and clinical utility not only to be both related and important issues, but to be complementary when it comes to implementing (or envisioning the possible implementation of) any automated system in real-world health care scenarios. Indeed, both these dimensions (How biased will the final system be? How clinically useful?) depend greatly on methodological choices spread throughout the development of the system.</p><p>Building on these considerations, we propose to study bias and clinical utility as 2 intricate dimensions rooted in methodological choices, with expected downstream social impact on users. To our knowledge, this is the first such extensive, large-scale joint analysis of bias and clinical utility for predictive mental health applications based on LLMs.</p></sec><sec id="s1-2"><title>Objectives</title><p>The objectives of this work are 2-fold. First, we wish to broadly describe the methodological decisions adopted in publications on LLM-based mental health prediction. This will lead us to analyze possible bias and barriers to clinical utility stemming from these methodological choices, at each step of the system&#x2019;s development pipeline.</p><p>Second, we intend to document the awareness of researchers on matters of bias and clinical utility, as indicated by relevant reported considerations in these same publications. Indeed, increased global research attention on these issues does not necessarily translate into actionable plans for other researchers [<xref ref-type="bibr" rid="ref32">32</xref>]. Furthermore, propensity to bias is still absent from reference evaluation benchmarks used to evaluate these systems, which shows limited appropriation of these issues within the community.</p><p>Drawing on these insights, we mean to foster discussions as to the extent to which it is currently possible, safe, and desirable to integrate LLMs into clinical routine and mental health care in particular.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Registration and Protocol</title><p>This scoping review was preregistered in the Open Science Framework registry [<xref ref-type="bibr" rid="ref33">33</xref>]. We followed the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews) guidelines [<xref ref-type="bibr" rid="ref34">34</xref>] (refer to <xref ref-type="supplementary-material" rid="app5">Checklist 1</xref>).</p></sec><sec id="s2-2"><title>Eligibility Criteria</title><p>We analyzed original research articles published in conferences or journals between January 2019 (to account for the release of BERT [<xref ref-type="bibr" rid="ref35">35</xref>]) and December 2024. Eligible studies had to (1) assess the presence or severity of a mental health condition; (2) use, for that purpose, an NLP system comprising at least one LLM; and (3) use nonsynthetic textual samples as input data, possibly along with synthetic samples or other modalities. We excluded preprints, reviews, and studies which were not openly available online, as well as papers not written in English.</p><p>Importantly, following criterion (1), articles claiming to perform only stress detection or sentiment analysis were not considered eligible. Regarding criterion (2), as it can be underspecified in the literature which models fall under the designation of &#x201C;LLMs,&#x201D; we followed the recommendation of Rogers and Luccioni [<xref ref-type="bibr" rid="ref36">36</xref>] to provide an explicit working definition. In accordance with standard considerations [<xref ref-type="bibr" rid="ref37">37</xref>], we therefore called LLMs NLP tools that (1) model and can generate text, (2) receive large-scale pretraining, (3) can be used for transfer learning, and (4) rely on the Transformer architecture [<xref ref-type="bibr" rid="ref38">38</xref>]. Notably, this definition includes both encoder models such as BERT [<xref ref-type="bibr" rid="ref35">35</xref>] and generative decoder models such as GPT [<xref ref-type="bibr" rid="ref39">39</xref>], in line with existing reviews about applications of LLMs to mental health [<xref ref-type="bibr" rid="ref24">24</xref>]. Finally, although criterion (3) allows for the inclusion of papers developing systems for audio or imaging data, our analysis was oriented toward the LLM-relevant part of these systems, that is, the one processing text. Besides, even if synthetic datasets are increasingly praised for clinical research [<xref ref-type="bibr" rid="ref40">40</xref>], we wanted to focus in this review on systems that fulfill at least partially expectations of pending deployment&#x2014;generally comprising testing in realistic conditions which are, up to now, only imperfectly proxied by generated corpora [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. In addition, our framework comprises a step which focuses on the included data, considering demographics of included participants, which is not applicable to synthetic datasets.</p></sec><sec id="s2-3"><title>Search Strategy</title><p>In order to provide a wide coverage of relevant publications, we searched 5 databases spanning biomedical, NLP, and more general AI-related literature: MEDLINE (PubMed), Web of Science, IEEE Xplore, ACM Digital Library, and the ACL Anthology. All searches were performed on January 20, 2025, with queries following the template (terms related to &#x201C;MENTAL HEALTH&#x201D;) AND (terms related to &#x201C;LARGE LANGUAGE MODELS&#x201D;) AND (terms related to &#x201C;PREDICTION&#x201D;). The queries were crafted following multiple search rounds and designed to minimize false negatives; hence, the large number of allowed terms in each keyword group (request details are presented in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). When allowed by the database search engine, we used filters for date, publication language, and publication type according to the eligibility criteria listed above.</p></sec><sec id="s2-4"><title>Screening Process</title><p>The initial set of records identified via database search was loaded into the Rayyan software (Mourad Ouzzani) [<xref ref-type="bibr" rid="ref43">43</xref>]. A sample of 100 records was first screened based on title and abstract by 3 reviewers (CB, AN, and KF) independently. Interannotator agreement was high, with pairwise Cohen &#x03BA; [<xref ref-type="bibr" rid="ref44">44</xref>] scores ranging from 0.654 to 0.807 [<xref ref-type="bibr" rid="ref45">45</xref>], while resulting discussions helped clarify the screening strategy (see &#x201C;Eligibility Criteria&#x201D; section). Subsequently, the first author (CB) screened the remainder of records, as well as the full text of all the eligible studies before conducting extraction. Exclusion reasons and record counts at every step are detailed in the flowchart. Remaining doubts were addressed through discussion between authors.</p></sec><sec id="s2-5"><title>Data Extraction</title><p>In accordance with our research objectives, we extracted 2 types of information in the reviewed papers: methodological information (what do the authors do?), and authors&#x2019; self-reflections and acknowledgments linked with bias and clinical utility (what do the authors say about what they do?). The full-text PDFs of included studies were loaded into Zotero (Corporation for Digital Scholarship) [<xref ref-type="bibr" rid="ref46">46</xref>] for reading, alongside an auxiliary spreadsheet document for qualitative entity extraction performed by CB.</p><p>To our knowledge, there is no existing framework specifically designed to study both bias and clinical utility in the conception pipelines for LLMs applied to mental health. However, 2 previous approaches, which we deem complementary to analyze bias and clinical utility at every step of the development pipeline of an LLM-based system for mental health prediction, were proposed.</p><p>On the one hand, Hovy and Prabhumoye [<xref ref-type="bibr" rid="ref29">29</xref>] identified 5 sources of bias in NLP systems: the data, the annotations made on the data, the input representations fed to NLP models, NLP models themselves, and larger research design choices such as the treated language. On the other hand, Chen et al [<xref ref-type="bibr" rid="ref30">30</xref>] sketch a 5-step pipeline of ethical stakes for machine learning (ML) in health care, considering (1) the unequal distribution of funding and research attention between health issues with regards to the share of the world population that they affect (problem selection), (2) nonrepresentative data (data collection), (3) imperfect medical proxies for target outcomes (outcome definition), (4) ML models optimization parameters potentially exacerbating bias (algorithmic development), and (5) blindness to issues such as generalizability to various clinical settings, downstream impact assessment or regulatory compliance before real-life system deployment (postdeployment considerations).</p><p>Inspired by these 2 previous approaches, we propose a 5-step framework (<xref ref-type="fig" rid="figure1">Figure 1</xref>) to study both bias and clinical utility within the selected articles. We identify key arguments of each work, and define qualitative entities to extract from papers for future analysis. For instance, Chen et al [<xref ref-type="bibr" rid="ref30">30</xref>] discuss the importance of measuring performance metrics across demographic groups to ensure fair treatment among patients, which we operationalize as a binary entity &#x201C;per-group performance&#x201D; to be set to &#x201C;true&#x201D; when authors report disaggregated results (eg, against sex or gender, race or ethnicity, or other sensitive attributes). This entity relates indeed both to bias (as some groups may be discriminated against if the system performs poorly for them) and to clinical utility (as clinical outcomes are expected to be fair among patients). We redirect the reader to <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> for more details on our analysis framework design and the operationalization of entities.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Left: our proposed 5-step pipeline for analyzing bias and clinical utility in large language model&#x2013;based mental health prediction systems. Right: entities extracted in papers, in accordance with each pipeline step. LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e88082_fig01.png"/></fig><p>We also collected sentences from the included articles that refer explicitly to any sort of bias or to elements related to the clinical utility of the developed system, as a proxy for self-reflection of researchers regarding these matters. We extracted these statements independently from the arguments they defend (eg, recognizing the presence of bias in presented results vs stating that bias has been mitigated), within our extraction framework.</p></sec><sec id="s2-6"><title>Data Analysis</title><p>We performed the analysis of the extracted data in a Jupyter Notebook (Project Jupyter) environment, using Python libraries (Guido van Rossum). When applicable, usual descriptive statistics (distribution of values, mean value, range, and so on) were computed. In some cases when necessary conditions were met, we modeled the effect of categorical variables (eg, the effect of medical affiliation on the studied conditions) using chi-squared contingency tests (<italic>&#x03C7;</italic><sup>2</sup>) computed with the <italic>SciPy</italic> Python package (without correction). The entities for which full sentences or noisy outputs were extracted were further analyzed by clustering the extracted data into meaningful categories, in an iterative and inductive aggregation process inspired by grounded theory [<xref ref-type="bibr" rid="ref47">47</xref>]. For authors&#x2019; self-reflections about bias and clinical utility, we used the same approach to map the highlighted sentences to themes. The final results (extracted entities or categories obtained after coding for each paper) following data analysis are presented as a spreadsheet in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Selection</title><p>We initially identified 2646 candidate records in databases (search performed on January 20, 2025). After deduplication, 2472 unique papers were screened for eligibility based on title and abstract. The remaining 263 papers were then screened based on full-text read. Finally, 201 (8.1%) studies were included in this review. The PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) diagram of the process is detailed in <xref ref-type="fig" rid="figure2">Figure 2</xref>, and the full list of included papers is available in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flowchart.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e88082_fig02.png"/></fig><p>Included studies (N=201) are mostly recent: 149 (74.1%) are published in 2023&#x2010;2024, 46 (22.9%) in 2021&#x2010;2022, and only 6 (3%) in 2019&#x2010;2020. Most of the retrieved studies are sourced from IEEE Xplore (124/201, 61.7%) and PubMed (57/201, 28.4%). The ACL Anthology, the Web of Science, and the ACM Digital Library, respectively, raised 10 (5%), 7 (3.5%), and 3 (1.5%) papers.</p></sec><sec id="s3-2"><title>Global Trends in Bias and Clinical Utility Report by the Authors</title><p>We found mentions of bias and clinical utility-related themes in 164/201 (81.6%) papers. The pipeline step that triggered the most discussion is that of data collection (107/201, 53.2%), followed by model development (69/201, 34.3%), and problem selection and research design (60/201, 29.8%). Themes associated with outcome definition and postdeployment considerations were only evoked in, respectively, 45/201 (22.4%) and 31/201 (15.4%) papers. In 77/201 (38.3%) papers, the discussion was restricted to a single step of the pipeline. However, 46/201 (22.9%), 26/201 (12.9%), and 10/201 (5%) papers evoked themes associated with 2, 3, and 4 of these steps. A total of 5 (2.5%) papers mentioned considerations from all 5 steps of the pipeline [<xref ref-type="bibr" rid="ref48">48</xref>-<xref ref-type="bibr" rid="ref52">52</xref>].</p></sec><sec id="s3-3"><title>Bias and Clinical Utility Along the System Development Pipeline</title><sec id="s3-3-1"><title>Overview</title><p>In this section, we describe entities extracted from the included studies, as illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>. For each of the 5 steps of pipeline design potentially subject to biases and clinical utility limitations, we both report the extracted entities and the relevant themes explicitly mentioned by authors about their work under a paragraph entitled &#x201C;Mentions of Bias and Clinical Utility&#x2013;Related Themes.&#x201D;</p></sec><sec id="s3-3-2"><title>Step 1: Research Design and Problem Selection</title><sec id="s3-3-2-1"><title>Overview</title><p>The step of research design and problem selection raises the following questions: who did the research? On which mental health conditions? To help whom?</p></sec><sec id="s3-3-2-2"><title>Affiliation Countries of Authors</title><p>Forty-five distinct countries are represented in author affiliations, spanning all 6 continents. There is, however, a concentration of publications with author affiliations in China (43/201, 21.4%), the United States (42/201, 20.9%), and India (32/201, 15.9%). International collaborations are present in 54 papers (26.9%), with up to 4 different countries of affiliation in a single publication.</p></sec><sec id="s3-3-2-3"><title>Domain Affiliation of Authors</title><p>A total of 76 (37.8%) papers comprise at least one author declaring an affiliation related to the medical domain (domain author). The average ratio of domain authors over all authors within a paper is 0.22, with a median value of 0.0 (IQR 0.33) and an SD of 0.34. A total of 20 (10%) papers have a domain-to-all ratio of 1.0, indicating that they are written only by domain-affiliated authors.</p></sec><sec id="s3-3-2-4"><title>Languages</title><p>As for the languages the authors work with (ie, those of the processed data), 18 distinct languages are identified (<xref ref-type="fig" rid="figure3">Figure 3</xref>). While English is overtly predominant (153/201, 76.1%), other treated languages include Chinese (29/201, 14.4%), Arabic (7/201, 3.5%), Thai (4/201, 2%), Portuguese (3/201, 1.5%), Japanese (3/201, 1.5%), and others. Of the publications on English and Chinese, respectively, 75/153 (49%) and 4/29 (13.8%) omit to state the treated language explicitly; therefore, this information has to be deduced based on the models or resources used.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Distribution of the languages studied in the included papers (ie, the languages of the data the systems make predictions about). For each language, it is also indicated whether it is explicitly mentioned by the authors in the publications.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e88082_fig03.png"/></fig></sec><sec id="s3-3-2-5"><title>Studied Mental Health Conditions</title><p>Fifteen mental health disorder groups (mapped to condition groups found in the <italic>DSM-5</italic> [<italic>Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition</italic>] [<xref ref-type="bibr" rid="ref53">53</xref>], as well as a generic category of &#x201C;Suicidal Risk&#x201D; for papers studying suicidal ideation, behavior, or attempts) are studied in the selected articles (<xref ref-type="table" rid="table1">Table 1</xref>). A majority of the reviewed papers work on predicting depressive disorders (148/201, 73.6%), significantly more so when the papers comprise no domain author (<italic>&#x03C7;</italic><sup>2</sup><sub>1</sub>=10.95, <italic>P</italic>&#x003C;.001). Other prevalent selected conditions include suicidal risk (47/201, 22.9%) and anxiety disorders (17/201, 8.5%), with a variety of other condition groups marginally represented. Besides, while 175 (87.1%) papers focus on a single condition, 26 (12.9%) papers study at least 2 of them, with a maximum coverage of 10 conditions in a single paper [<xref ref-type="bibr" rid="ref54">54</xref>].</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Distribution of mental health disorder groups among studies (some studies include multiple disorder groups).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Mental health disorder group</td><td align="left" valign="bottom">Paper count, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Depressive disorders</td><td align="left" valign="top">148 (73.6)</td></tr><tr><td align="left" valign="top">Suicidal risk</td><td align="left" valign="top">47 (23.4)</td></tr><tr><td align="left" valign="top">Anxiety disorders</td><td align="left" valign="top">17 (8.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Trauma and stressor-related disorders</td><td align="left" valign="top">14 (7)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Bipolar and related disorders</td><td align="left" valign="top">13 (6.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Schizophrenia spectrum and other psychiatric disorders</td><td align="left" valign="top">11 (5.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Attention-deficit hyperactivity disorder</td><td align="left" valign="top">10 (5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Personality disorders</td><td align="left" valign="top">8 (4)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Feeding and eating disorders</td><td align="left" valign="top">7 (3.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Major and mild neurocognitive disorders</td><td align="left" valign="top">5 (2.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Autism spectrum disorders</td><td align="left" valign="top">5 (2.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Intellectual disabilities</td><td align="left" valign="top">3 (1.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Obsessive-compulsive and related disorders</td><td align="left" valign="top">2 (1)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Dissociative disorders</td><td align="left" valign="top">1 (0.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Other (perinatal psychiatry)</td><td align="left" valign="top">1 (0.5)</td><td align="left" valign="top"/></tr></tbody></table></table-wrap></sec><sec id="s3-3-2-6"><title>Rationale for Problem Selection</title><p>Three major themes are identified which relate to the conditions under study or to mental health in general: negative impacts on individuals (165/201, 82.1%; eg, &#x201C;Mental illness is [...] significantly impacting the lives of individuals across diverse communities&#x201D; [<xref ref-type="bibr" rid="ref54">54</xref>]), a high or increasing prevalence (170/201, 84.6%; eg, &#x201C;Suicide is one of the main causes of death in the world&#x201D; [<xref ref-type="bibr" rid="ref55">55</xref>]), and shortcomings of the &#x201C;traditional&#x201D; medical system to efficiently tackle the problem (127/201, 63.2%; eg, &#x201C;Constrained clinician time and a strong focus on anticancer treatment may contribute to the insufficient identification of patients at risk for depression&#x201D; [<xref ref-type="bibr" rid="ref56">56</xref>]). Besides, 9/201 (4.5%) papers explain that their study takes place in the context of shared tasks organized by the scientific community. These tasks are hosted by labs or workshops at conferences such as CLEF, CLPsych, ACL, and IEEE BigData.</p></sec><sec id="s3-3-2-7"><title>Mentions of Bias and Clinical Utility&#x2013;Related Themes</title><p>Overall, 60/201 (29.8%) papers mention themes related to the step of research design and problem selection, and 38/201 (18.9%) mention language and culture as a potential bias source related to the step of problem selection. This comprises both acknowledgments that results on English may not generalize to other languages, and cases where authors highlight that they work on an underserved language. Others (22/201, 10.9%) explicate how population disparities related to the condition under study oriented the design of their work toward more vulnerable groups. For instance, some researchers study mental health issues of pregnant or postpartum women [<xref ref-type="bibr" rid="ref49">49</xref>], US service members and veterans [<xref ref-type="bibr" rid="ref57">57</xref>], children or adolescents [<xref ref-type="bibr" rid="ref58">58</xref>], patients with cancer [<xref ref-type="bibr" rid="ref59">59</xref>], and so on. As for the choice of a condition, 4 (6.7%) papers claim to tackle a research gap, such as Juhng et al [<xref ref-type="bibr" rid="ref52">52</xref>] who state that &#x201C;much less work in the NLP community has focused on detecting anxiety disorders as has been done for depressive disorders.&#x201D; Finally, 1 (0.5%) paper explicitly acknowledges the &#x201C;absence of direct involvement with mental health experts,&#x201D; which we map to the theme of workforce diversity.</p></sec></sec></sec><sec id="s3-4"><title>Step 2: Data Collection</title><sec id="s3-4-1"><title>Overview</title><p>The step of data collection raises the following questions: who is represented in the data, and how? How was the data collected?</p></sec><sec id="s3-4-2"><title>Data Sources</title><p>We identify 3 major data provenances for the datasets used in the reviewed papers. First, posts from social media and online forums are used in 133/201 (66.2%) papers, under the frequent rationale that it consists of large-scale, spontaneous, and anonymous testimonies of users about their mental health. This data source is significantly more adopted in papers comprising no domain authors (<italic>&#x03C7;</italic><sup>2</sup><sub>1</sub>=23.57, <italic>P</italic>&#x003C;.001). The most prevalent sources within this category are Reddit (Reddit, Inc; 72/201, 35.8%), Twitter (X Corp; 44/201, 21.9%), and Sina Weibo (Weibo Corporation; 11/201, 5.5%). Data collections such as those provided in the CLEF conference for eRisk labs constitute a popular example of such social media datasets (see <xref ref-type="other" rid="box1">Textbox 1</xref>). Second, 62/201 (30.8%) papers make use of data collected from participants during clinical or semiclinical interactions. This comprises interview transcripts (eg, Patient Health Questionnaire (PHQ)&#x2013;led interviews in the distressed analysis interview corpus and extended-distressed analysis interview corpus [<xref ref-type="bibr" rid="ref60">60</xref>], refer to <xref ref-type="other" rid="box1">Textbox 1</xref>), narrative transcripts provided by participants about their medical experiences, texts entered on mental health monitoring apps, and therapy transcripts. Third, electronic health records of patients coming from different sites are used in 10/201 (4.9%) papers. Among them, one paper uses publicly available ScAN [<xref ref-type="bibr" rid="ref61">61</xref>] (from MIMIC-III [<xref ref-type="bibr" rid="ref62">62</xref>]), while the others rely on private datasets accessed through their institutional affiliations. A total of 2 (1%) papers gather multisite records, while the others are from a single source.</p><boxed-text id="box1"><title> Erisk and the Distressed Analysis Interview Corpus (DAIC): 2 common datasets for automated mental health prediction.</title><p>Erisk labs have been held at CLEF since 2017. Each edition comprises one or more tasks designed to &#x201C;explore issues of evaluation methodology, effectiveness metrics, and other processes related to early risk detection&#x201D; [<xref ref-type="bibr" rid="ref63">63</xref>]. As of 2025, while multiple tasks have been proposed (anorexia and related eating disorders, self-harm, pathological gambling), the major focus has been on depression detection. Notably:</p><list list-type="bullet"><list-item><p>A first dataset for the early detection of depression was used and extended throughout 2017 (task 1), 2018 (task 1), and 2022 (task 2) editions [<xref ref-type="bibr" rid="ref63">63</xref>-<xref ref-type="bibr" rid="ref65">65</xref>]. It initially consisted of 531,000 Reddit (Reddit, Inc) posts [<xref ref-type="bibr" rid="ref66">66</xref>] collected from 892 Reddit users (137 depressed and 755 control) following a template-based heuristic and manual check. &#x201C;Depression&#x201D; posts are those of users who unambiguously stated their diagnosis (eg, &#x201C;I was diagnosed with depression,&#x201D; but not &#x201C;I think I have depression&#x201D;), while &#x201C;Control&#x201D; posts were those of random users as well as some posting in the r/Depression forum without explicit diagnosis statements.</p></list-item><list-item><p>A second dataset was constituted for measuring the severity of the signs of depression in 2019 (task 3) and extended for 2020 (task 2) and 2021 (task 3) [<xref ref-type="bibr" rid="ref67">67</xref>-<xref ref-type="bibr" rid="ref69">69</xref>]. It contains the entire Reddit posting history of 20 users (up to 90 in 2021) who agreed to fill BDI-based questionnaires as ground-truth data. Lab participants thus have to predict the users&#x2019; scores for each item, as well as an overall depression assessment score, eventually mapped to 1 of 4 coarse-grained categories (minimal, mild, moderate, and severe depression).</p></list-item></list><p>The DAIC is a collection of 621 clinical interviews designed to &#x201C;support the diagnosis of psychological distress conditions such as anxiety, depression, and post traumatic stress disorder&#x201D; [<xref ref-type="bibr" rid="ref60">60</xref>]. It comprises 4 subcorpora corresponding to distinct interview setups: face-to-face, teleconference, Wizard-of-Oz (WOZ, ie, interviews are led by a human-controlled virtual interviewer), and automated (ie, interviews are led by a fully automated virtual agent). The corpus data consist of audio and video recordings of interviews, partial transcriptions, and both verbal and nonverbal annotations. Interviewees consisted of veterans of the US armed forces and voluntary participants recruited via online ads in California, 397/621 (63.9%) of whom had their interview tagged as &#x201C;distressed.&#x201D; An extended version (E-DAIC) has since been constituted for the AVEC workshop at ACM MM 2019, although documentation is still lacking [<xref ref-type="bibr" rid="ref70">70</xref>].</p></boxed-text></sec><sec id="s3-4-3"><title>Reporting of Participants Count and Demographic Characteristics</title><p>We find that 122/201 (60.7%) papers omit to provide the number of participants included in the datasets they use in their work. When that count is disclosed, we find both very small (eg, 22 participants in the study by Li et al [<xref ref-type="bibr" rid="ref71">71</xref>]) and very large datasets (eg, more than 43,000 patients in the study by Meng et al [<xref ref-type="bibr" rid="ref72">72</xref>]). Demographic information about participants is reported in 43/201 (21.4%) papers. More precisely, information can be found about their age (35/201, 17.4%), their sex or gender (32/201, 15.9%), their race or ethnicity or cultural background (11/201, 5.5%), their education level or academic status (9/201, 4.5%), and other attributes such as marital status, household income, or additional medical information (9/201, 4.5%).</p></sec><sec id="s3-4-4"><title>Reporting of Documents Count and Other Dataset Characteristics</title><p>Document counts (ie, the data samples which are fed to the model, such as social media posts or clinical notes) are reported in 172/201 (85.6%) papers. We note that this document count ranges from dozens of documents (eg, the study by Hayati et al [<xref ref-type="bibr" rid="ref73">73</xref>]) to several hundred thousand documents (eg, the study by Feng et al [<xref ref-type="bibr" rid="ref74">74</xref>]). Details about the time of data collection, class size (eg, the number of data samples tagged with or without depression), or outcome-related statistics (eg, the average anxiety score in a questionnaire-based dataset) are reported by the authors in 127/201 (63.2%) papers.</p></sec><sec id="s3-4-5"><title>Mentions of Bias and Clinical Utility&#x2013;Related Themes</title><p>We find mentions related to the step of data collection in 107/201 (53.2%) papers, making it the most discussed pipeline step with regard to bias and clinical utility. A total of 60/201 (29.8%) studies discuss the challenge of class imbalance (ie, when target labels are unevenly distributed in the data), which can lead to biased results. This imbalance is typically mitigated by authors using sampling or weighting techniques to avoid negative impacts on model performance. Representation bias mentions (ie, when the attributes of the people represented in a dataset mismatch those of the target population) are found in 52/201 (25.9%) papers and linked with diverse methodological (eg, site of collection and sampling method) and sociodemographic features (eg, gender, race and ethnicity, and social status). Other themes relate to the challenges of dealing with a small dataset (22/201, 10.9%), preexisting bias (3/201, 1.5%) in the data (eg, Hutto et al [<xref ref-type="bibr" rid="ref51">51</xref>] acknowledge as a limitation that &#x201C;bias may be introduced by the author of the [medical] note&#x201D; and static datasets (3/201, 2.8%) which can only account for a person&#x2019;s condition at a given point in time, whereas clinical decisions are usually made based on a person&#x2019;s medical history.</p></sec></sec><sec id="s3-5"><title>Step 3: Outcome Definition</title><sec id="s3-5-1"><title>Overview</title><p>The step of outcome definition raises the following questions: what is the expected outcome? How does it translate to clinically actionable information?</p></sec><sec id="s3-5-2"><title>Outcome Modalities</title><p>Different outcome modalities are found in the reviewed studies, with sometimes several ones in the same paper. In 146/201 (72.6%) papers, simple binary labels are assigned to data samples for the condition under study (eg, &#x201C;depression&#x201D; vs &#x201C;no depression&#x201D;). In 46/201 (22.8%) papers, severity degrees are instead estimated on ordinal, qualitative scales typically comprised 3 to 4 values (eg, &#x201C;no depression,&#x201D; &#x201C;low depression,&#x201D; &#x201C;moderate depression,&#x201D; and &#x201C;severe depression&#x201D;). Continuous, quantitative scales, on the other hand, are used in 19/201 (9.5%) papers when scores are computed based on reference questionnaires (see paragraph below). Finally, 9/201 (4.5%) papers provide symptom-level predictions by focusing on individual items of aforementioned questionnaires, emotions, or other behavioral clues.</p></sec><sec id="s3-5-3"><title>Reference Diagnostic Tools or Methods Used</title><p>Only 64/201 (31.8%) articles explicitly mention their reliance on standard psychiatric tools and classifications for condition assessment of the included participants. A total of 10/201 (5%) studies refer to the <italic>DSM</italic> (<italic>Diagnostic and Statistical Manual of Mental Disorders</italic>), and 11/201 (5.5%) to the <italic>ICD</italic> (<italic>International Classification of Diseases</italic>). A range of questionnaires are also mentioned, including the PHQ (30/201, 14.9%) with 8 (PHQ-8) or 9 items (PHQ-9) [<xref ref-type="bibr" rid="ref75">75</xref>], the Hamilton Depression Scale [<xref ref-type="bibr" rid="ref76">76</xref>] (6/201, 3%) and the Beck Depression Inventory [<xref ref-type="bibr" rid="ref77">77</xref>] (6/201, 3%) for depression, as well as the posttraumatic stress disorder checklist for civilians [<xref ref-type="bibr" rid="ref78">78</xref>] (3/201, 1.5%), or the Mini-Mental State Examination [<xref ref-type="bibr" rid="ref79">79</xref>] (3/201, 1.5%) for cognitive impairment. It should be noted that the authors use such questionnaires in various ways: sometimes, scores obtained from participants&#x2019; self-administration are used as readily available labels, while in other cases the questionnaire is used to structure interviews whose transcripts will be used as input to the prediction system. In 137/201 (68.2%) studies, the authors rely on other custom methods for condition identification, or do not specify the means by which their data were annotated. These custom methods can involve asking annotators to classify texts based on the presence of certain keywords or on social media&#x2013;based heuristics (eg, when texts are extracted from specific discussion forums). In 17/201 (8.5%) studies, the authors specify that labels were provided by trained clinicians.</p></sec><sec id="s3-5-4"><title>Mentions of Bias and Clinical Utility&#x2013;Related Themes</title><p>Mentions related to the step of outcome definition are found in 45/201 (22.4%) papers. The validity of the proxy used as a marker of a given mental health condition is questioned in 38/201 (18.9%) papers, which stress the limitations of relying on binary labels or self-report statements expressed on social media. For instance, Ohse et al [<xref ref-type="bibr" rid="ref80">80</xref>] note that the absence of a differential diagnosis may limit the validity of their findings (derived from self-report measures), as their ground truth may have been overly inclusive. Annotation bias is also mentioned in 8/201 (4%) papers, which comprises acknowledgments that low interannotator agreement or stereotypical bias from the annotators may hinder the validity of the labels used.</p></sec></sec><sec id="s3-6"><title>Step 4: Model Development</title><sec id="s3-6-1"><title>Overview</title><p>The step of model development raises the following questions: which LLMs are used? How are they evaluated?</p></sec><sec id="s3-6-2"><title>LLMs Used</title><p>A total of 112 distinct LLMs are identified in our corpus, with an average count of 2.5 LLMs used per study (range: 1&#x2010;14). As detailed in <xref ref-type="table" rid="table2">Table 2</xref>, most studies rely on encoder-only models mainly derived from BERT [<xref ref-type="bibr" rid="ref35">35</xref>] and related models such as RoBERTa [<xref ref-type="bibr" rid="ref81">81</xref>], DistilBERT [<xref ref-type="bibr" rid="ref82">82</xref>], and others. These models are typically used to produce contextual representations of input sequences, which are later fed to classification layers or to other models for mental health prediction. Decoder-based models derived from GPT [<xref ref-type="bibr" rid="ref39">39</xref>] and XLNet [<xref ref-type="bibr" rid="ref83">83</xref>] as well as encoder-decoder models are used to a lesser extent; researchers then tend to use them for prediction in a few-shot setting or, more rarely, for data augmentation or explanation generation. In 38/201 (18.9%) papers, at least one LLM which was previously trained on clinical or mental health&#x2013;related content is used. We note, however, that in the remaining 163/201 (81.1%) papers, general models are preferred, which can be used off-the-shelf and optionally trained on domain data. When working on languages other than English, authors explicitly state the need to use multilingual or specialized monolingual models such as BERT derivatives for the Chinese or Arabic language.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>List of frequently used models in our corpus (with cutoff frequency of 4 papers)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Clinical or mental health training data</td><td align="left" valign="bottom">Paper count, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">BERT<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">124 (61.7)</td></tr><tr><td align="left" valign="top">RoBERTa [<xref ref-type="bibr" rid="ref81">81</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">49 (24.4)</td></tr><tr><td align="left" valign="top">DistilBERT [<xref ref-type="bibr" rid="ref82">82</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">25 (12.4)</td></tr><tr><td align="left" valign="top">MentalBERT [<xref ref-type="bibr" rid="ref84">84</xref>]</td><td align="left" valign="top">Reddit mental health posts</td><td align="left" valign="top">21 (10.4)</td></tr><tr><td align="left" valign="top">ALBERT [<xref ref-type="bibr" rid="ref85">85</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">15 (7.5)</td></tr><tr><td align="left" valign="top">GPT-3.5</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">15 (7.5)</td></tr><tr><td align="left" valign="top">XLNet [<xref ref-type="bibr" rid="ref83">83</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">15 (7.5)</td></tr><tr><td align="left" valign="top">GPT-4 [<xref ref-type="bibr" rid="ref86">86</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">10 (5)</td></tr><tr><td align="left" valign="top">GPT or ChatGPT</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">10 (5)</td></tr><tr><td align="left" valign="top">XLM-RoBERTa [<xref ref-type="bibr" rid="ref87">87</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">9 (4.5)</td></tr><tr><td align="left" valign="top">BERT-chinese [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">9 (4.5)</td></tr><tr><td align="left" valign="top">MentalRoBERTa [<xref ref-type="bibr" rid="ref84">84</xref>]</td><td align="left" valign="top">Reddit mental health posts</td><td align="left" valign="top">7 (3.5)</td></tr><tr><td align="left" valign="top">SBERT [<xref ref-type="bibr" rid="ref88">88</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">6 (3)</td></tr><tr><td align="left" valign="top">mBERT [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">6 (3)</td></tr><tr><td align="left" valign="top">BioClinicalBERT [<xref ref-type="bibr" rid="ref89">89</xref>]</td><td align="left" valign="top">Clinical notes (from MIMIC-III)</td><td align="left" valign="top">5 (2.5)</td></tr><tr><td align="left" valign="top">MPNET [<xref ref-type="bibr" rid="ref90">90</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">5 (2.5)</td></tr><tr><td align="left" valign="top">ELECTRA [<xref ref-type="bibr" rid="ref91">91</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">5 (2.5)</td></tr><tr><td align="left" valign="top">DeBERTa [<xref ref-type="bibr" rid="ref92">92</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">5 (2.5)</td></tr><tr><td align="left" valign="top">AraBERT [<xref ref-type="bibr" rid="ref93">93</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">4 (2)</td></tr><tr><td align="left" valign="top">BioBERT [<xref ref-type="bibr" rid="ref94">94</xref>]</td><td align="left" valign="top">Biomedical literature (PubMed, PMC)</td><td align="left" valign="top">4 (2)</td></tr><tr><td align="left" valign="top">DepRoBERTa [<xref ref-type="bibr" rid="ref95">95</xref>]</td><td align="left" valign="top">Reddit mental health posts</td><td align="left" valign="top">4 (2)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Indicates cases where the precise version of the GPT model used is not disclosed.</p></fn><fn id="table2fn2"><p><sup>b</sup>BERT: Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table2fn3"><p><sup>c</sup>No clinical or mental health training data have been performed for these models.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6-3"><title>Reported Metrics and Human Evaluation</title><p>Standard classification metrics such as accuracy, precision, specificity, recall (sensitivity), F-measure, area under the receiver operating characteristic curve, or raw confusion matrices are reported in 187/201 (93%) papers. Similarly, standard regression metrics such as mean squared error, mean absolute error, or correlation coefficients (eg, Pearson <italic>r</italic>) are reported in 25/201 (12.4%) papers. A range of more specific metrics used to fit the constraints of particular prediction setups was also identified: for instance, the early risk detection error [<xref ref-type="bibr" rid="ref67">67</xref>] is designed to take into account both the correctness of a system&#x2019;s binary decision and the delay needed to make that decision (measured by the number of text items seen before providing an answer). On the other hand, human-led, qualitative evaluation is very rare in the corpus. As an example, Wang et al [<xref ref-type="bibr" rid="ref96">96</xref>] ask 50 medical interns specializing in mental illness to rate the safety, usability, and fluency of their depression detection system on a 1&#x2010;10 scale.</p></sec><sec id="s3-6-4"><title>Mentions of Bias and Clinical Utility Related Themes</title><p>Mentions related to the step of model development are found in 69/201 (34.3%) papers. The main reported theme is that of the technical limitations of models (56/201, 27.8%), which can have an effect on their downstream clinical utility, although this consequence is hardly ever made explicit by the authors. These limitations include small context window size (making the processing of long documents challenging), nondomain training, an elevated computational and time cost, the need for large training datasets, etc. In addition, 13/201 (6.5%) papers explicitly mention the notion of model bias, that is, bias intrinsically encoded and amplified by LLMs. Group fairness is evoked in 5/201 (2.5%) papers, some of which report disaggregated results based on sociodemographic features (eg, age and gender [<xref ref-type="bibr" rid="ref97">97</xref>] and sex, race, and ethnicity [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref56">56</xref>]).</p></sec></sec><sec id="s3-7"><title>Step 5: Postdeployment Considerations</title><sec id="s3-7-1"><title>Overview</title><p>The step of postdeployment considerations raises the following questions: which measures are implemented to ensure a fair deployment? For which use cases?</p></sec><sec id="s3-7-2"><title>Intended Use Cases</title><p>Many papers do not propose an explicit, concrete use case for their work. Instead, some express the general ambition to facilitate the early detection of mental health issues in individuals (57/201, 28.4%). As for more precise applications, some researchers propose to use clues in patients&#x2019; medical data to anticipate the possible onset of mental health issues (eg, depression in patients with cancer [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref56">56</xref>]). Others evoke the use of NLP methods to monitor global mental health trends in social media (eg, postpartum depression [<xref ref-type="bibr" rid="ref98">98</xref>] and suicide ideation [<xref ref-type="bibr" rid="ref99">99</xref>]), sometimes even suggesting that users at risk should be recommended to mental health specialists [<xref ref-type="bibr" rid="ref59">59</xref>] or identified for real-world interventions [<xref ref-type="bibr" rid="ref100">100</xref>]. Some proposals are also designed to fit more directly into the patient-caregiver relationship. Notably, Diniz et al [<xref ref-type="bibr" rid="ref55">55</xref>] developed a web app for doctors to monitor suicidal ideations of patients estimated from their smartphone keyboard data, while Shimamoto et al [<xref ref-type="bibr" rid="ref101">101</xref>] proposed to automatically estimate depression severity based on oral responses to an automated version of the Montgomery-&#x00C5;sberg Depression Rating Scale.</p></sec><sec id="s3-7-3"><title>Validation in Clinical Settings Procedures</title><p>The corpus contains no description of rigorous, systematic validation procedures in realistic clinical settings of the reviewed automated systems for mental health prediction. This could be explained by the fact that a large number of studies are retrospective, that is, they use existing data from patients the authors did not interact directly with. In that case, there is consequently no downstream impact on the concerned stakeholders, and the clinical utility is minimal. To our knowledge, none of the reviewed systems has been deployed in routine mental health care.</p></sec><sec id="s3-7-4"><title>Mentions of Bias and Clinical Utility&#x2013;Related Themes</title><p>Mentions related to the step of postdeployment considerations are found in 31/201 (15.4%) papers. A total of 26/201 (12.9%) papers evoke the performance of their system in the context of a future possible deployment; both positive and negative judgments are emitted. As an illustration, Matero et al [<xref ref-type="bibr" rid="ref102">102</xref>] explicitly warn that &#x201C;at this time [they] do not suggest [their] model(s) be used in practice to label mental health states,&#x201D; whereas Bartal et al [<xref ref-type="bibr" rid="ref50">50</xref>] are more confident that their model &#x201C;has the potential to fit seamlessly into routine obstetric care.&#x201D; Others anticipate whether the generalizability (4/201, 2%) of their system is sufficient or whether it may be confronted with integration challenges (4/201, 2.0%). Interestingly, Khalil et al [<xref ref-type="bibr" rid="ref103">103</xref>] suggest that federated learning (ie, &#x201C;an alternative that leaves the training data distributed on the mobile devices, and learns a shared model by aggregating locally-computed updates&#x201D; [<xref ref-type="bibr" rid="ref104">104</xref>]) could address issues of data privacy and regulation disparities between countries when making mental health predictions in a multilingual setting.</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>With this study, we introduced a framework to analyze methodological bias and clinical utility throughout the development pipeline of LLM-based mental health prediction systems. Our review notably sketched the prototypical profile of such a prediction system: it focuses on depressive disorders, leverages English data collected from social media, and uses BERT-based classifiers in the absence of a clinically useful outcome for the included participants. In what follows, we argue that this is problematic for multiple reasons.</p></sec><sec id="s4-2"><title>LLM-Based Mental Health Prediction Suffers From Bias All Along the Development Pipeline</title><p>First, the focus on depressive disorders leaves nearly untouched other condition families, echoing the observations of Wang et al [<xref ref-type="bibr" rid="ref26">26</xref>]. Although it is true that they account for a large part of the world&#x2019;s mental health burden, millions of individuals are also affected by anxiety disorders, bipolar disorder, schizophrenia, or eating disorders, to name a few [<xref ref-type="bibr" rid="ref105">105</xref>]. Besides, it seems that communication disorders (affecting individuals&#x2019; language abilities) could also particularly benefit from NLP inputs [<xref ref-type="bibr" rid="ref106">106</xref>]. In addition, despite a relative diversity among affiliation countries, papers almost exclusively work on English data, which exacerbates the existing bias toward English-based systems in NLP [<xref ref-type="bibr" rid="ref29">29</xref>]. This bias often goes unnoticed, with nearly half of papers working on English omitting to mention it explicitly, a phenomenon that has been otherwise measured in NLP conferences [<xref ref-type="bibr" rid="ref107">107</xref>,<xref ref-type="bibr" rid="ref108">108</xref>]. In order to benefit a wider range of individuals, non-English or multilingual approaches should be considered.</p><p>Second, the data used in the studies largely come from social media (which was also observed by Wang et al [<xref ref-type="bibr" rid="ref26">26</xref>]), even more so when no domain authors are involved. Some authors highlight the ease of access and collection of such data (as opposed to &#x201C;traditional&#x201D; clinical data), its alleged relevance to the task of mental health detection, and its wide accessibility (eg, &#x201C;social media&#x2019;s ubiquity presents a platform for individuals to express their feelings, instead of traditional, formal clinical settings, with 8 out of 10 people disclosing their suicidal thoughts and plans.&#x201D; [<xref ref-type="bibr" rid="ref109">109</xref>]). However, social media users do not make up representative samples of individuals, and it is difficult to ensure the quality of social media posts [<xref ref-type="bibr" rid="ref30">30</xref>]. This is furthermore problematic as demographic information is generally absent from these datasets, and users are rarely able to provide consent for the collection of their data. In order to avoid harming users, social media research should comply with ethical guidelines as regards consent of participants, data deidentification, and sharing [<xref ref-type="bibr" rid="ref110">110</xref>].</p><p>Third, the proxies used to account for participants&#x2019; alleged mental health status lack clinical grounding. It is unlikely that heuristics based on keywords within texts or self-administered questionnaire scores would be considered reasonable indicators by clinicians to validate diagnostic labels. This leads us to question the maturity of the field for clinical integration.</p><p>Fourth, authors tend to overlook practical requirements when adopting a model. We note that popular, nonspecialist models (and, increasingly, generative LLMs) are generally preferred, yet these choices are rarely clinically motivated. It is currently debated whether specialist models can actually be competitive with bigger, more recent generalist models [<xref ref-type="bibr" rid="ref111">111</xref>] on specialized downstream tasks. However, the size and eventual proprietary status of the latter hinder nonnegotiable needs of data privacy and control over computational cost in clinical environments with limited resources, along with issues of environmental impact and models&#x2019; propensity to bias [<xref ref-type="bibr" rid="ref112">112</xref>,<xref ref-type="bibr" rid="ref113">113</xref>].</p><p>Finally, there is an overall lack of awareness of the ethical issues implied by a possible clinical deployment of the considered systems. Importantly, some works seem not to aim for such clinical use cases and deny that their systems should be used as diagnostic tools already. Yet this leads to questioning the clinical utility of tasks confined to computer science laboratories, whose results may be mistakenly interpreted as clinical conclusions. Overall, authors mainly initiate discussion themes that focus on generic ML issues related to model constraints at the model development step (eg, input length constraints, computational cost seen as a practical barrier and not an environmental problem), while revealing a data-centric view of bias around the data collection step (eg, calling for bigger, class-balanced datasets). This technical approach to bias, which calls for technical solutions, has been previously identified as the engineering ethos [<xref ref-type="bibr" rid="ref114">114</xref>]. However, such complex issues as biases cannot be solved with a sole technique [<xref ref-type="bibr" rid="ref12">12</xref>], if at all [<xref ref-type="bibr" rid="ref18">18</xref>]. In what follows, we argue that an ethical approach to LLMs applied to the domain of mental health would benefit from more interdisciplinarity.</p></sec><sec id="s4-3"><title>Interdisciplinarity Is Needed to Increase Clinical Validity and Utility</title><p>In addition to the biases identified above, our observations raise the question of interdisciplinarity and its influence on the clinical utility of proposed solutions. Indeed, while we stressed global shortcomings, it should be acknowledged that studies were more robust in terms of disorder diversity, specificity of disorder definition, and quality of data sources (clinical interviews and electronic health records) when at least one author had a medical affiliation. This aligns with previous findings that constituting interdisciplinary teams (including clinicians and computer scientists) improves results validity in ML problems [<xref ref-type="bibr" rid="ref115">115</xref>].</p><p>For instance, clinical expertise is needed to adequately define what is referred to as &#x201C;depression&#x201D; in studies. Some papers ambiguously assimilate the concepts of &#x201C;depression&#x201D; and &#x201C;negative mood/sentiment&#x201D; (eg, &#x201C;the task is to discover the mood of the user&#x201D; [<xref ref-type="bibr" rid="ref116">116</xref>]), or &#x201C;depression&#x201D; and &#x201C;suicidal risk&#x201D; (eg, Wang et al [<xref ref-type="bibr" rid="ref117">117</xref>] guidelines for depression level estimation are actually based on reports of suicidal ideation and plan), or do not seem to make a conceptual difference between self-diagnosed depression and clinical diagnoses produced by health care professionals. The task of disambiguating what is designated by &#x201C;depression&#x201D; is all the more complex because according to MeSH definitions, it can cover a sign or symptom (&#x201C;Depression&#x201D; [<xref ref-type="bibr" rid="ref118">118</xref>]) or a disorder with different levels of severity (&#x201C;Depressive Disorder&#x201D; [<xref ref-type="bibr" rid="ref119">119</xref>] or &#x201C;Major Depressive Disorder&#x201D; [<xref ref-type="bibr" rid="ref120">120</xref>]). These differences are rarely described or implemented in data annotation protocols in studies, even though they play a crucial role in diagnostic reasoning and patient care. Consequently, training a system to automatically detect &#x201C;depression&#x201D; and claiming it is ready for use in clinical practice is suboptimal at best (what kind of &#x201C;depression&#x201D; does the system detect?), and misleading if users are left to make their own assumption of what concept of &#x201C;depression&#x201D; is operationalized. The close collaboration between computer scientists and medical doctors is thus required to ensure and to propagate the clinical value of the targeted outcome.</p><p>Interdisciplinary collaborations between diverse fields can also improve the robustness of systems&#x2019; evaluation. Indeed, while some studies warrant caution before real-world deployment, others present their system as deployment-ready (even in the absence of such validation), while a majority do not address the question at all. Notably, no included study presented a rigorous clinical validation for an LLM-based prediction system to be applied in practice, for example, through a randomized controlled trial [<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>Extending even beyond sole clinical validation, interdisciplinary frameworks for the validation of digital devices in medicine have recently been published. For instance, the certification of digital devices&#x2014;and particularly those embarking LLMs&#x2014;by the US Food and Drug Administration or the European Medicines Agency requires them to comply with the criteria of the V3 framework [<xref ref-type="bibr" rid="ref121">121</xref>]. This framework requires rigorous clinical validation (ie, demonstrating that the metrics computed by the device correlate with meaningful physiological or clinical dimensions), but also rigorous hardware verification (validation of the sensor, if there is one) and analytical validation (ie, demonstrating that the algorithm indeed measures a behavior or a physiological marker).</p><p>This validation, however, does not ensure that the device, in its context of use, will be useful. Indeed, for most of the articles included in this review, the motivation for the research often remains vague, although untested potential use cases are described, such as large-scale mental health screening or patient monitoring. For others, very few contextual elements help grasp the medical stakes, which we believe have to be mentioned in such research. Is this to say that mental health prediction could become yet another LLM-equipped task, ensuring reasonable chances of getting a paper published?</p><p>Interestingly, an updated version of the V3 framework has been recently introduced: the V3+ framework, completing the 3 previous validations by a criterion of usability validation [<xref ref-type="bibr" rid="ref122">122</xref>]. Although it differs from clinical utility, the addition of utility validation incorporates considerations regarding the projected use, the target population, and the intended context into the validation of digital health devices. Clarifying these elements, which requires interdisciplinary consultation between computer scientists (regarding technical validity) and clinicians (regarding usage), would thus prevent misunderstandings between the two communities, thereby avoiding empty promises fueled by each community&#x2019;s overclaims regarding the capabilities of the other&#x2014;particularly regarding LLMs&#x2019; ability in clinical context [<xref ref-type="bibr" rid="ref123">123</xref>].</p></sec><sec id="s4-4"><title>Limitations</title><p>While this scoping review focuses on 201 papers published between 2019 and 2024, this period covers both the introduction of masked (eg, BERT) and autoregressive (eg, GPT) language models, which provides an opportunity to observe how they were used for mental health prediction early on. In addition, we believe that the large number of included studies allows us to provide a representative snapshot of the field in this 6-year window, through an original joint analysis of bias and clinical utility. While more recent articles could have been included, we assume that they would only have had a marginal effect on the trends we evidenced. Notably, Reiter [<xref ref-type="bibr" rid="ref20">20</xref>] recently reported that as of March 2025, only 0.1% of papers from the ACL Anthology provided any form of downstream impact evaluation, echoing our findings in the case of mental health prediction.</p><p>Our approach is that of a scoping review, aiming at presenting a representative view of what researchers or clinicians may encounter when looking for LLM-based mental health prediction works. As such, we did not perform quality assessment of the articles at the screening stage [<xref ref-type="bibr" rid="ref124">124</xref>]. In addition, due to the large number of included studies, we could not ensure an independent extraction by multiple reviewers, which may introduce bias in the presented results: as for other phases of the review, this was mitigated with regular discussions and methodological refinements. Since we considered a large number of qualitative entities, methodological choices were necessary to extract them into quantitative metrics. For instance, authors were labeled as &#x201C;domain authors&#x201D; whenever they reported being affiliated with a medical institution, a medical university department or school, or a company providing medical devices or services. While we acknowledge this to be an imperfect proxy for medical expertise, our data are freely available in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendices 1</xref><xref ref-type="supplementary-material" rid="app2"/><xref ref-type="supplementary-material" rid="app3"/>-<xref ref-type="supplementary-material" rid="app4">4</xref> for interested researchers to replicate our results with different methodological choices.</p><p>Finally, this study was conducted by a group of French, White, NLP researchers, including some with a background in medical informatics or psychiatry. Although we have aimed at adopting an open and interdisciplinary vision, taking into account issues related to bias and clinical utility, we may have missed some relevant aspects to these questions, starting with the phase of article identification. Because this is the main language used for research dissemination, we reviewed only articles written in English in major scientific databases, which nonetheless allowed us to retrieve studies working on other languages. We intentionally crafted broad queries to minimize false negatives, and considered multiple literature sources relevant to NLP and medical sciences. Yet, different requests and additional sources may retrieve slightly different results. Also, information about eventual clinical deployments of the considered systems outside of what is reported in the papers may have been missed.</p></sec><sec id="s4-5"><title>Broader Implications</title><p>This review focused solely on LLM-based mental health prediction. We did not consider interventional applications (eg, chatbots for mental health support), nor other medical domains, and some of our observations may not generalize. Nevertheless, we believe that some of our findings are part of a broader picture. Our study suggests that over the past 6 years, the perspective of downstream usage (and users) has seldom been considered in the development of LLM-based, predictive mental health applications. This echoes trends that have been reported in NLP [<xref ref-type="bibr" rid="ref20">20</xref>]. The overemphasis on algorithm development [<xref ref-type="bibr" rid="ref125">125</xref>] and quantitative, benchmark-based competition for performance [<xref ref-type="bibr" rid="ref126">126</xref>] poses a risk of harm by assuming medical applications are &#x201C;yet another task&#x201D; to tackle despite increased ethical risks. Finally, it would be interesting to study more deeply the &#x201C;promises&#x201D; and narratives around AI and NLP for (mental) health care (eg, the promise to assist health care workers without replacing them, the promise to be cost-effective, and so on), which is left for future research.</p></sec><sec id="s4-6"><title>Conclusions</title><p>LLMs are increasingly used in mental health prediction tasks. Following a 5-step pipeline as a framework to analyze NLP proposals, we described an emerging field that exhibits some marked tendencies, notably toward the detection of depressive disorders and the use of social media data in English. We showed that researchers are generally aware of some bias and clinical utility&#x2013;related issues, but that reports in papers are sparse and not systematic. In particular, postdeployment considerations and impact studies are lacking, which leads us to question the idea that LLMs have the potential to revolutionize mental health care. When dealing with NLP-assisted mental health research, we advocate for a combined view of bias and clinical utility, implying that no biased system can be clinically useful, and vice versa. This requires interdisciplinarity and careful efforts all along the research pipeline. We strongly believe that the goal to contribute to mental health research and, ultimately, to benefit patients should remain central in NLP efforts towards mental health.</p></sec></sec></body><back><ack><p>The authors thank Ga&#x00E9;tan Kerdelhu&#x00E9; from D&#x00E9;SaN (Digital Health Department, University Hospital of Rouen, France) for his guidance in crafting MEDLINE queries and selecting an appropriate review management platform. The authors attest that no AI was used to assist the creation of the manuscript, nor the underlying research work.</p></ack><notes><sec><title>Funding</title><p>This work has received support from the French government (Agence Nationale pour la Recherche) under grant agreement ANR-23-IAS1-0004 (InExtenso).</p></sec><sec><title>Data Availability</title><p>All data generated or analyzed during this study are included in this published article and its supplementary information files.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: CB, KF, AN</p><p>Data curation: CB, KF, AN</p><p>Formal analysis: CB</p><p>Funding acquisition: KF</p><p>Investigation: CB, VPM</p><p>Methodology: CB, KF, AN</p><p>Supervision: KF, VPM, AN</p><p>Visualization: CB</p><p>Writing - Original Draft: CB, KF, VPM, AN</p><p>Writing - Review &#x0026; Editing: CB, KF, VPM, AN</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb2"><italic>DSM</italic></term><def><p><italic>Diagnostic and Statistical Manual of Mental Disorders</italic></p></def></def-item><def-item><term id="abb3"><italic>DSM-5</italic></term><def><p><italic>Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition</italic></p></def></def-item><def-item><term id="abb4"><italic>ICD</italic></term><def><p><italic>International Classification of Diseases</italic></p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb7">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb8">PHQ</term><def><p>Patient Health Questionnaire</p></def></def-item><def-item><term id="abb9">PHQ-8</term><def><p>Patient Health Questionnaire-8</p></def></def-item><def-item><term id="abb10">PHQ-9</term><def><p>Patient Health Questionnaire-9</p></def></def-item><def-item><term id="abb11">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb12">PRISMA-ScR</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>K</given-names> </name><name name-style="western"><surname>Mao</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>A survey of large language models for healthcare: from data, technology, and applications to accountability and ethics</article-title><source>Inform Fusion</source><year>2025</year><month>06</month><volume>118</volume><fpage>102963</fpage><pub-id pub-id-type="doi">10.1016/j.inffus.2025.102963</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nazi</surname><given-names>ZA</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>W</given-names> </name></person-group><article-title>Large language models in healthcare and medical domain: a review</article-title><source>Informatics</source><year>2024</year><month>09</month><volume>11</volume><issue>3</issue><fpage>57</fpage><pub-id pub-id-type="doi">10.3390/informatics11030057</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>T&#x00F8;lb&#x00F8;ll</surname><given-names>K</given-names> </name></person-group><article-title>Linguistic features in depression: a meta-analysis</article-title><source>J Lang Works</source><year>2019</year><month>12</month><day>16</day><access-date>2026-07-27</access-date><volume>4</volume><issue>2</issue><fpage>39</fpage><comment><ext-link ext-link-type="uri" xlink:href="https://tidsskrift.dk/lwo/article/view/117798">https://tidsskrift.dk/lwo/article/view/117798</ext-link></comment></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Spoletini</surname><given-names>I</given-names> </name><name name-style="western"><surname>Rubino</surname><given-names>IA</given-names> </name><etal/></person-group><article-title>The language of schizophrenia: an analysis of micro and macrolinguistic abilities and their neuropsychological correlates</article-title><source>Schizophr Res</source><year>2008</year><month>10</month><volume>105</volume><issue>1-3</issue><fpage>144</fpage><lpage>155</lpage><pub-id pub-id-type="doi">10.1016/j.schres.2008.07.011</pub-id><pub-id pub-id-type="medline">18768300</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Amblard</surname><given-names>M</given-names> </name><name name-style="western"><surname>Musiol</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rebuschi</surname><given-names>M</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Amblard</surname><given-names>M</given-names> </name><name name-style="western"><surname>Musiol</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rebuschi</surname><given-names>M</given-names> </name></person-group><article-title>Discourse coherence - from psychology to linguistics and back again</article-title><source>Coherence of Discourse - Formal and Conceptual Issues of Language</source><year>2021</year><publisher-name>Springer</publisher-name><fpage>1</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-71434-5_1</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Appell</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kertesz</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fisman</surname><given-names>M</given-names> </name></person-group><article-title>A study of language functioning in Alzheimer patients</article-title><source>Brain Lang</source><year>1982</year><month>09</month><volume>17</volume><issue>1</issue><fpage>73</fpage><lpage>91</lpage><pub-id pub-id-type="doi">10.1016/0093-934x(82)90006-2</pub-id><pub-id pub-id-type="medline">7139272</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Demner-Fushman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chapman</surname><given-names>WW</given-names> </name><name name-style="western"><surname>McDonald</surname><given-names>CJ</given-names> </name></person-group><article-title>What can natural language processing do for clinical decision support?</article-title><source>J Biomed Inform</source><year>2009</year><month>10</month><volume>42</volume><issue>5</issue><fpage>760</fpage><lpage>772</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2009.08.007</pub-id><pub-id pub-id-type="medline">19683066</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fraser</surname><given-names>KC</given-names> </name><name name-style="western"><surname>Meltzer</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Rudzicz</surname><given-names>F</given-names> </name></person-group><article-title>Linguistic features identify Alzheimer&#x2019;s disease in narrative speech</article-title><source>J Alzheimers Dis</source><year>2016</year><volume>49</volume><issue>2</issue><fpage>407</fpage><lpage>422</lpage><pub-id pub-id-type="doi">10.3233/JAD-150520</pub-id><pub-id pub-id-type="medline">26484921</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leroy</surname><given-names>G</given-names> </name><name name-style="western"><surname>Gu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Pettygrove</surname><given-names>S</given-names> </name><name name-style="western"><surname>Galindo</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Arora</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kurzius-Spencer</surname><given-names>M</given-names> </name></person-group><article-title>Automated extraction of diagnostic criteria from electronic health records for autism spectrum disorders: development, evaluation, and application</article-title><source>J Med Internet Res</source><year>2018</year><month>11</month><day>7</day><volume>20</volume><issue>11</issue><fpage>e10497</fpage><pub-id pub-id-type="doi">10.2196/10497</pub-id><pub-id pub-id-type="medline">30404767</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hiebel</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ferret</surname><given-names>O</given-names> </name><name name-style="western"><surname>Fort</surname><given-names>K</given-names> </name><name name-style="western"><surname>N&#x00E9;v&#x00E9;ol</surname><given-names>A</given-names> </name></person-group><article-title>Clinical text generation: are we there yet?</article-title><source>Annu Rev Biomed Data Sci</source><year>2025</year><month>08</month><volume>8</volume><issue>1</issue><fpage>173</fpage><lpage>198</lpage><pub-id pub-id-type="doi">10.1146/annurev-biodatasci-103123-095202</pub-id><pub-id pub-id-type="medline">40101215</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>H</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>F</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Z</given-names> </name></person-group><article-title>A survey of recent methods for addressing AI fairness and bias in biomedicine</article-title><source>J Biomed Inform</source><year>2024</year><month>06</month><volume>154</volume><fpage>104646</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2024.104646</pub-id><pub-id pub-id-type="medline">38677633</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hofmann</surname><given-names>V</given-names> </name><name name-style="western"><surname>Kalluri</surname><given-names>PR</given-names> </name><name name-style="western"><surname>Jurafsky</surname><given-names>D</given-names> </name><name name-style="western"><surname>King</surname><given-names>S</given-names> </name></person-group><article-title>AI generates covertly racist decisions about people based on their dialect</article-title><source>Nature</source><year>2024</year><month>09</month><volume>633</volume><issue>8028</issue><fpage>147</fpage><lpage>154</lpage><pub-id pub-id-type="doi">10.1038/s41586-024-07856-5</pub-id><pub-id pub-id-type="medline">39198640</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zack</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lehman</surname><given-names>E</given-names> </name><name name-style="western"><surname>Suzgun</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Assessing the potential of GPT-4 to perpetuate racial and gender biases in health care: a model evaluation study</article-title><source>Lancet Digit Health</source><year>2024</year><month>01</month><volume>6</volume><issue>1</issue><fpage>e12</fpage><lpage>e22</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00225-X</pub-id><pub-id pub-id-type="medline">38123252</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ducel</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hiebel</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ferret</surname><given-names>O</given-names> </name><name name-style="western"><surname>Fort</surname><given-names>K</given-names> </name><name name-style="western"><surname>N&#x00E9;v&#x00E9;ol</surname><given-names>A</given-names> </name></person-group><article-title>&#x201C;Women do not have heart attacks!&#x201D; gender biases in automatically generated clinical cases in French</article-title><access-date>2026-08-05</access-date><conf-name>Findings of the Association for Computational Linguistics</conf-name><conf-date>Apr 29 to May 4, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.findings-naacl">https://aclanthology.org/2025.findings-naacl</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>P Goddu</surname><given-names>A</given-names> </name><name name-style="western"><surname>O&#x2019;Conor</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Lanzkron</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Do words matter? Stigmatizing language and the transmission of bias in the medical record</article-title><source>J Gen Intern Med</source><year>2018</year><month>05</month><volume>33</volume><issue>5</issue><fpage>685</fpage><lpage>691</lpage><pub-id pub-id-type="doi">10.1007/s11606-017-4289-2</pub-id><pub-id pub-id-type="medline">29374357</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>L</given-names> </name><name name-style="western"><surname>Mao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Hofmann</surname><given-names>V</given-names> </name><name name-style="western"><surname>Bai</surname><given-names>X</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Che</surname><given-names>W</given-names> </name><name name-style="western"><surname>Nabende</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shutova</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pilehvar</surname><given-names>MT</given-names> </name></person-group><article-title>Aligned but blind: alignment increases implicit bias by reducing awareness of race</article-title><conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><conf-loc>Vienna, Austria</conf-loc><fpage>22167</fpage><lpage>22184</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.1078</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Omar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Soffer</surname><given-names>S</given-names> </name><name name-style="western"><surname>Agbareia</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Sociodemographic biases in medical decision making by large language models</article-title><source>Nat Med</source><year>2025</year><month>06</month><volume>31</volume><issue>6</issue><fpage>1873</fpage><lpage>1881</lpage><pub-id pub-id-type="doi">10.1038/s41591-025-03626-6</pub-id><pub-id pub-id-type="medline">40195448</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Resnik</surname><given-names>P</given-names> </name></person-group><article-title>Large language models are biased because they are large language models</article-title><source>Comput Linguist</source><year>2025</year><month>09</month><day>1</day><volume>51</volume><issue>3</issue><fpage>885</fpage><lpage>906</lpage><pub-id pub-id-type="doi">10.1162/coli_a_00558</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Bhanushali</surname><given-names>T</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Badami</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hightow-Weidman</surname><given-names>L</given-names> </name></person-group><article-title>Evaluating generative AI in mental health: systematic review of capabilities and limitations</article-title><source>JMIR Ment Health</source><year>2025</year><month>05</month><day>15</day><volume>12</volume><fpage>e70014</fpage><pub-id pub-id-type="doi">10.2196/70014</pub-id><pub-id pub-id-type="medline">40373033</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reiter</surname><given-names>E</given-names> </name></person-group><article-title>We should evaluate real-world impact</article-title><source>Comput Linguist</source><year>2025</year><month>12</month><day>1</day><volume>51</volume><issue>4</issue><fpage>1419</fpage><lpage>1431</lpage><pub-id pub-id-type="doi">10.1162/COLI.a.18</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ong</surname><given-names>JCL</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>SYH</given-names> </name><name name-style="western"><surname>William</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Ethical and regulatory challenges of large language models in medicine</article-title><source>Lancet Digit Health</source><year>2024</year><month>06</month><volume>6</volume><issue>6</issue><fpage>e428</fpage><lpage>e432</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(24)00061-X</pub-id><pub-id pub-id-type="medline">38658283</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Badrick</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bowling</surname><given-names>F</given-names> </name></person-group><article-title>Clinical utility - information about the usefulness of tests</article-title><source>Clin Biochem</source><year>2023</year><month>11</month><volume>121-122</volume><fpage>110656</fpage><pub-id pub-id-type="doi">10.1016/j.clinbiochem.2023.110656</pub-id><pub-id pub-id-type="medline">37802380</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moulaei</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yadegari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Baharestani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Farzanbakhsh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sabet</surname><given-names>B</given-names> </name><name name-style="western"><surname>Reza Afrash</surname><given-names>M</given-names> </name></person-group><article-title>Generative artificial intelligence in healthcare: a scoping review on benefits, challenges and applications</article-title><source>Int J Med Inform</source><year>2024</year><month>08</month><volume>188</volume><fpage>105474</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2024.105474</pub-id><pub-id pub-id-type="medline">38733640</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Thygesen</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Farrington</surname><given-names>J</given-names> </name><name name-style="western"><surname>Keen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name></person-group><article-title>Large language models for mental health applications: systematic review</article-title><source>JMIR Ment Health</source><year>2024</year><month>10</month><day>18</day><volume>11</volume><issue>1</issue><fpage>e57400</fpage><pub-id pub-id-type="doi">10.2196/57400</pub-id><pub-id pub-id-type="medline">39423368</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>P</given-names> </name><etal/></person-group><article-title>The applications of large language models in mental health: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>5</day><volume>27</volume><fpage>e69284</fpage><pub-id pub-id-type="doi">10.2196/69284</pub-id><pub-id pub-id-type="medline">40324177</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>G</given-names> </name></person-group><article-title>The application and ethical implication of generative AI in mental health: systematic review</article-title><source>JMIR Ment Health</source><year>2025</year><month>06</month><day>27</day><volume>12</volume><fpage>e70610</fpage><pub-id pub-id-type="doi">10.2196/70610</pub-id><pub-id pub-id-type="medline">40577783</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bedi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Orr-Ewing</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Testing and evaluation of health care applications of large language models: a systematic review</article-title><source>JAMA</source><year>2025</year><month>01</month><day>28</day><volume>333</volume><issue>4</issue><fpage>319</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1001/jama.2024.21700</pub-id><pub-id pub-id-type="medline">39405325</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schnepper</surname><given-names>R</given-names> </name><name name-style="western"><surname>Roemmel</surname><given-names>N</given-names> </name><name name-style="western"><surname>Schaefert</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lambrecht-Walzinger</surname><given-names>L</given-names> </name><name name-style="western"><surname>Meinlschmidt</surname><given-names>G</given-names> </name></person-group><article-title>Exploring biases of large language models in the field of mental health: comparative questionnaire study of the effect of gender and sexual orientation in anorexia nervosa and bulimia nervosa case vignettes</article-title><source>JMIR Ment Health</source><year>2025</year><month>03</month><day>20</day><volume>12</volume><issue>1</issue><fpage>e57986</fpage><pub-id pub-id-type="doi">10.2196/57986</pub-id><pub-id pub-id-type="medline">40111287</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hovy</surname><given-names>D</given-names> </name><name name-style="western"><surname>Prabhumoye</surname><given-names>S</given-names> </name></person-group><article-title>Five sources of bias in natural language processing</article-title><source>Lang Linguist Compass</source><year>2021</year><month>08</month><volume>15</volume><issue>8</issue><fpage>e12432</fpage><pub-id pub-id-type="doi">10.1111/lnc3.12432</pub-id><pub-id pub-id-type="medline">35864931</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>IY</given-names> </name><name name-style="western"><surname>Pierson</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rose</surname><given-names>S</given-names> </name><name name-style="western"><surname>Joshi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ferryman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ghassemi</surname><given-names>M</given-names> </name></person-group><article-title>Ethical machine learning in healthcare</article-title><source>Annu Rev Biomed Data Sci</source><year>2021</year><month>07</month><volume>4</volume><fpage>123</fpage><lpage>144</lpage><pub-id pub-id-type="doi">10.1146/annurev-biodatasci-092820-114757</pub-id><pub-id pub-id-type="medline">34396058</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wenderott</surname><given-names>K</given-names> </name><name name-style="western"><surname>Krups</surname><given-names>J</given-names> </name><name name-style="western"><surname>Weigl</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wooldridge</surname><given-names>AR</given-names> </name></person-group><article-title>Facilitators and barriers to implementing AI in routine medical imaging: systematic review and qualitative analysis</article-title><source>J Med Internet Res</source><year>2025</year><month>07</month><day>21</day><volume>27</volume><fpage>e63649</fpage><pub-id pub-id-type="doi">10.2196/63649</pub-id><pub-id pub-id-type="medline">40690758</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ghosh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wilson</surname><given-names>K</given-names> </name></person-group><article-title>Bias Is a math problem, AI bias is a technical problem: 10-year literature review of AI/LLM bias research reveals narrow [gender-centric] conceptions of &#x2018;Bias&#x2019;, and academia-industry gap</article-title><source>AIES</source><year>2025</year><month>10</month><day>15</day><volume>8</volume><issue>2</issue><fpage>1091</fpage><lpage>1106</lpage><pub-id pub-id-type="doi">10.1609/aies.v8i2.36613</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="web"><article-title>Large language models for mental health diagnosis: a scoping review of biases and applicability concerns</article-title><source>OSF</source><year>2025</year><access-date>2025-11-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://osf.io/ygknh/overview?view_only=aac52b70d01f4bcb994ab1533c0bbc74">https://osf.io/ygknh/overview?view_only=aac52b70d01f4bcb994ab1533c0bbc74</ext-link></comment></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tricco</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Lillie</surname><given-names>E</given-names> </name><name name-style="western"><surname>Zarin</surname><given-names>W</given-names> </name><etal/></person-group><article-title>PRISMA Extension for Scoping Reviews (PRISMA-ScR): checklist and explanation</article-title><source>Ann Intern Med</source><year>2018</year><month>10</month><day>2</day><volume>169</volume><issue>7</issue><fpage>467</fpage><lpage>473</lpage><pub-id pub-id-type="doi">10.7326/M18-0850</pub-id><pub-id pub-id-type="medline">30178033</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Devlin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>MW</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>K</given-names> </name><name name-style="western"><surname>Toutanova</surname><given-names>K</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Burstein</surname><given-names>J</given-names> </name><name name-style="western"><surname>Doran</surname><given-names>C</given-names> </name><name name-style="western"><surname>Solorio</surname><given-names>T</given-names> </name></person-group><article-title>BERT: pre-training of deep bidirectional transformers for language understanding</article-title><conf-name>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)</conf-name><conf-date>Jun 2-9, 2019</conf-date><conf-loc>Minneapolis, MN</conf-loc><fpage>4171</fpage><lpage>4186</lpage><pub-id pub-id-type="doi">10.18653/v1/N19-1423</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Rogers</surname><given-names>A</given-names> </name><name name-style="western"><surname>Luccioni</surname><given-names>S</given-names> </name></person-group><article-title>Position: key claims in LLM research have a long tail of footnotes</article-title><conf-name>Proceedings of the Forty-first International Conference on Machine Learning</conf-name><conf-date>Jul 21-27, 2024</conf-date><conf-loc>Vienna, Austria</conf-loc><fpage>42647</fpage><lpage>42466</lpage><pub-id pub-id-type="doi">10.5555/3692070.3693805</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Jurafsky</surname><given-names>DH</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>J</given-names> </name></person-group><source>Speech and Language Processing</source><year>2026</year><access-date>2026-07-27</access-date><edition>3</edition><publisher-name>Stanford University</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://web.stanford.edu/~jurafsky/slp3/">https://web.stanford.edu/~jurafsky/slp3/</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Vaswani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Parmar</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Attention is all you need</article-title><access-date>2026-07-27</access-date><conf-name>Annual Conference on Neural Information Processing Systems</conf-name><conf-date>Dec 4-9, 2017</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2017/hash/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html">https://proceedings.neurips.cc/paper/2017/hash/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Radford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Narasimhan</surname><given-names>K</given-names> </name></person-group><article-title>Improving language understanding by generative pre-training</article-title><source>Semantic Scholar</source><year>2018</year><access-date>2026-07-27</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.semanticscholar.org/paper/Improving-Language-Understanding-by-Generative-Radford-Narasimhan/cd18800a0fe0b668a1cc19f2ec95b5003d0a5035">https://www.semanticscholar.org/paper/Improving-Language-Understanding-by-Generative-Radford-Narasimhan/cd18800a0fe0b668a1cc19f2ec95b5003d0a5035</ext-link></comment></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Smolyak</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bjarnad&#x00F3;ttir</surname><given-names>MV</given-names> </name><name name-style="western"><surname>Crowley</surname><given-names>K</given-names> </name><name name-style="western"><surname>Agarwal</surname><given-names>R</given-names> </name></person-group><article-title>Large language models and synthetic health data: progress and prospects</article-title><source>JAMIA Open</source><year>2024</year><month>12</month><volume>7</volume><issue>4</issue><fpage>ooae114</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooae114</pub-id><pub-id pub-id-type="medline">39464796</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sharoff</surname><given-names>S</given-names> </name><name name-style="western"><surname>Baker</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hunt</surname><given-names>DDF</given-names> </name><name name-style="western"><surname>Simpson</surname><given-names>A</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Danilova</surname><given-names>V</given-names> </name><name name-style="western"><surname>Kurfal&#x0131;</surname><given-names>M</given-names> </name><name name-style="western"><surname>S&#x00F6;derfeldt</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Reed</surname><given-names>J</given-names> </name><name name-style="western"><surname>Burchell</surname><given-names>A</given-names> </name></person-group><article-title>Almost clinical: linguistic properties of synthetic electronic health records</article-title><conf-name>Proceedings of the 1st Workshop on Linguistic Analysis for Health (HeaLing 2026)</conf-name><conf-date>Mar 28, 2026</conf-date><conf-loc>Rabat, Morocco</conf-loc><fpage>115</fpage><lpage>126</lpage><pub-id pub-id-type="doi">10.18653/v1/2026.healing-1.10</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kapania</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ballard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kessler</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vaughan</surname><given-names>JW</given-names> </name></person-group><article-title>Examining the expanding role of synthetic data throughout the AI development pipeline</article-title><year>2025</year><month>06</month><day>23</day><conf-name>Proceedings of the 2025 ACM Conference on Fairness, Accountability, and Transparency</conf-name><conf-date>Jun 23-26, 2025</conf-date><conf-loc>Athens Greece</conf-loc><fpage>45</fpage><lpage>60</lpage><pub-id pub-id-type="doi">10.1145/3715275.3732005</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ouzzani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hammady</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fedorowicz</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Elmagarmid</surname><given-names>A</given-names> </name></person-group><article-title>Rayyan-a web and mobile app for systematic reviews</article-title><source>Syst Rev</source><year>2016</year><month>12</month><day>5</day><volume>5</volume><issue>1</issue><fpage>210</fpage><pub-id pub-id-type="doi">10.1186/s13643-016-0384-4</pub-id><pub-id pub-id-type="medline">27919275</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name></person-group><article-title>A coefficient of agreement for nominal scales</article-title><source>Educ Psychol Meas</source><year>1960</year><month>04</month><volume>20</volume><issue>1</issue><fpage>37</fpage><lpage>46</lpage><pub-id pub-id-type="doi">10.1177/001316446002000104</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Landis</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Koch</surname><given-names>GG</given-names> </name></person-group><article-title>The measurement of observer agreement for categorical data</article-title><source>Biometrics</source><year>1977</year><month>03</month><volume>33</volume><issue>1</issue><fpage>159</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.2307/2529310</pub-id><pub-id pub-id-type="medline">843571</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="web"><source>Zotero</source><access-date>2026-07-27</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.zotero.org/">https://www.zotero.org/</ext-link></comment></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Smith</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>DCP</given-names> </name><name name-style="western"><surname>Hsiao</surname><given-names>DK</given-names> </name></person-group><article-title>Database abstractions: aggregation and generalization</article-title><source>ACM Trans Database Syst</source><year>1977</year><volume>2</volume><issue>2</issue><fpage>105</fpage><lpage>133</lpage><pub-id pub-id-type="doi">10.1145/320544.320546</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van Buchem</surname><given-names>MM</given-names> </name><name name-style="western"><surname>de Hond</surname><given-names>AAH</given-names> </name><name name-style="western"><surname>Fanconi</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Applying natural language processing to patient messages to identify depression concerns in cancer patients</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>10</month><day>1</day><volume>31</volume><issue>10</issue><fpage>2255</fpage><lpage>2262</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae188</pub-id><pub-id pub-id-type="medline">39018490</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bartal</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jagodnik</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Babu</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Dekel</surname><given-names>S</given-names> </name></person-group><article-title>Identifying women with postdelivery posttraumatic stress disorder using natural language processing of personal childbirth narratives</article-title><source>Am J Obstet Gynecol MFM</source><year>2023</year><month>03</month><volume>5</volume><issue>3</issue><fpage>100834</fpage><pub-id pub-id-type="doi">10.1016/j.ajogmf.2022.100834</pub-id><pub-id pub-id-type="medline">36509356</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bartal</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jagodnik</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Dekel</surname><given-names>S</given-names> </name></person-group><article-title>AI and narrative embeddings detect PTSD following childbirth via birth stories</article-title><source>Sci Rep</source><year>2024</year><month>04</month><day>11</day><volume>14</volume><issue>1</issue><fpage>8336</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-54242-2</pub-id><pub-id pub-id-type="medline">38605073</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hutto</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zikry</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Bohac</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Using a natural language processing toolkit to classify electronic health records by psychiatric diagnosis</article-title><source>Health Informatics J</source><year>2024</year><volume>30</volume><issue>4</issue><fpage>14604582241296411</fpage><pub-id pub-id-type="doi">10.1177/14604582241296411</pub-id><pub-id pub-id-type="medline">39466373</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Juhng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Matero</surname><given-names>M</given-names> </name><name name-style="western"><surname>Varadarajan</surname><given-names>V</given-names> </name><name name-style="western"><surname>Eichstaedt</surname><given-names>J</given-names> </name><name name-style="western"><surname>V Ganesan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schwartz</surname><given-names>HA</given-names> </name></person-group><article-title>Discourse-level representations can improve prediction of degree of anxiety</article-title><conf-name>Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers)</conf-name><conf-date>Jul 9-14, 2023</conf-date><conf-loc>Toronto, Canada</conf-loc><fpage>1500</fpage><lpage>1511</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.acl-short.128</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="book"><source>Diagnostic and Statistical Manual of Mental Disorders: DSM-5TM</source><year>2013</year><edition>5</edition><publisher-name>American Psychiatric Association Publishing</publisher-name><pub-id pub-id-type="doi">10.1176/appi.books.9780890425596</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Adel</surname><given-names>S</given-names> </name><name name-style="western"><surname>Elmadany</surname><given-names>N</given-names> </name><name name-style="western"><surname>Sharkas</surname><given-names>M</given-names> </name></person-group><article-title>AI - driven mental disorders categorization from social media: a deep learning pre-screening framework</article-title><conf-name>2024 International Conference on Machine Intelligence and Smart Innovation (ICMISI)</conf-name><conf-date>May 12-14, 2024</conf-date><conf-loc>Alexandria, Egypt</conf-loc><fpage>238</fpage><lpage>243</lpage><pub-id pub-id-type="doi">10.1109/ICMISI61517.2024.10580665</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Diniz</surname><given-names>EJS</given-names> </name><name name-style="western"><surname>Fontenele</surname><given-names>JE</given-names> </name><name name-style="western"><surname>de Oliveira</surname><given-names>AC</given-names> </name><etal/></person-group><article-title>Boamente: a natural language processing-based digital phenotyping tool for smart monitoring of suicidal ideation</article-title><source>Healthcare (Basel)</source><year>2022</year><month>04</month><day>8</day><volume>10</volume><issue>4</issue><fpage>698</fpage><pub-id pub-id-type="doi">10.3390/healthcare10040698</pub-id><pub-id pub-id-type="medline">35455874</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Hond</surname><given-names>A</given-names> </name><name name-style="western"><surname>van Buchem</surname><given-names>M</given-names> </name><name name-style="western"><surname>Fanconi</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Predicting depression risk in patients with cancer using multimodal data: algorithm development study</article-title><source>JMIR Med Inform</source><year>2024</year><month>01</month><day>18</day><volume>12</volume><fpage>e51925</fpage><pub-id pub-id-type="doi">10.2196/51925</pub-id><pub-id pub-id-type="medline">38236635</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zuromski</surname><given-names>KL</given-names> </name><name name-style="western"><surname>Low</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>NC</given-names> </name><etal/></person-group><article-title>Detecting suicide risk among U.S. servicemembers and veterans: a deep learning approach using social media data</article-title><source>Psychol Med</source><year>2024</year><month>09</month><volume>54</volume><issue>12</issue><fpage>3379</fpage><lpage>3388</lpage><pub-id pub-id-type="doi">10.1017/S0033291724001557</pub-id><pub-id pub-id-type="medline">39245902</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Meng</surname><given-names>W</given-names> </name><name name-style="western"><surname>Hao</surname><given-names>J</given-names> </name><etal/></person-group><article-title>BERT- BiLSTM-Caps Language Model for Screening of Children&#x2019;s Severe Mental Retardation</article-title><conf-name>2021 20th International Conference on Ubiquitous Computing and Communications (IUCC/CIT/DSCI/SmartCNS)</conf-name><conf-date>Dec 20-22, 2021</conf-date><conf-loc>London, United Kingdom</conf-loc><fpage>296</fpage><lpage>301</lpage><pub-id pub-id-type="doi">10.1109/IUCC-CIT-DSCI-SmartCNS55181.2021.00056</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Podina</surname><given-names>IR</given-names> </name><name name-style="western"><surname>Bucur</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Todea</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Mental health at different stages of cancer survival: a natural language processing study of Reddit posts</article-title><source>Front Psychol</source><year>2023</year><volume>14</volume><fpage>1150227</fpage><pub-id pub-id-type="doi">10.3389/fpsyg.2023.1150227</pub-id><pub-id pub-id-type="medline">37425170</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Gratch</surname><given-names>J</given-names> </name><name name-style="western"><surname>Artstein</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lucas</surname><given-names>G</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Calzolari</surname><given-names>N</given-names> </name><name name-style="western"><surname>Choukri</surname><given-names>K</given-names> </name><name name-style="western"><surname>Declerck</surname><given-names>T</given-names> </name></person-group><article-title>The distress analysis interview corpus of human and computer interviews</article-title><conf-name>Ninth International Conference on Language Resources and Evaluation</conf-name><conf-date>May 26-31, 2014</conf-date><conf-loc>Reykjavik, Iceland</conf-loc><fpage>3123</fpage><lpage>3128</lpage><pub-id pub-id-type="doi">10.63317/3o7bccg9xequ</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Rawat</surname><given-names>BPS</given-names> </name><name name-style="western"><surname>Kovaly</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Pigeon</surname><given-names>W</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Carpuat</surname><given-names>M</given-names> </name><name name-style="western"><surname>Marneffe</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Ruiz</surname><given-names>M</given-names>  <suffix>IV</suffix></name></person-group><article-title>ScAN: suicide attempt and ideation events dataset</article-title><conf-name>Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics</conf-name><conf-date>Jul 10-15, 2022</conf-date><conf-loc>Seattle, WA</conf-loc><fpage>1029</fpage><lpage>1040</lpage><pub-id pub-id-type="doi">10.18653/v1/2022.naacl-main.75</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pollard</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mark</surname><given-names>R</given-names> </name></person-group><article-title>MIMIC-III clinical database</article-title><source>PhysioNet</source><access-date>2026-07-27</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://physionet.org/content/mimiciii/1.4/">https://physionet.org/content/mimiciii/1.4/</ext-link></comment></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Losada</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Crestani</surname><given-names>F</given-names> </name><name name-style="western"><surname>Parapar</surname><given-names>J</given-names> </name></person-group><article-title>ERISK 2017: CLEF lab on early risk prediction on the internet: experimental foundations</article-title><year>2017</year><conf-name>Experimental IR Meets Multilinguality, Multimodality, and Interaction (CLEF 2017)</conf-name><conf-date>Sep 11-14, 2017</conf-date><conf-loc>Dublin, Ireland</conf-loc><publisher-name>Springer International Publishing</publisher-name><fpage>346</fpage><lpage>360</lpage><pub-id pub-id-type="doi">10.1007/978-3-319-65813-1_30</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Losada</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Crestani</surname><given-names>F</given-names> </name><name name-style="western"><surname>Parapar</surname><given-names>J</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Bellot</surname><given-names>P</given-names> </name><name name-style="western"><surname>Trabelsi</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mothe</surname><given-names>J</given-names> </name></person-group><article-title>Overview of erisk: early risk prediction on the internet</article-title><source>Experimental IR Meets Multilinguality, Multimodality, and Interaction (CLEF 2018)</source><year>2018</year><publisher-name>Springer International Publishing</publisher-name><fpage>343</fpage><lpage>361</lpage><pub-id pub-id-type="doi">10.1007/978-3-319-98932-7_30</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Parapar</surname><given-names>J</given-names> </name><name name-style="western"><surname>Mart&#x00ED;n-Rodilla</surname><given-names>P</given-names> </name><name name-style="western"><surname>Losada</surname><given-names>DE</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Barr&#x00F3;n-Cede&#x00F1;o</surname><given-names>A</given-names> </name><name name-style="western"><surname>Da San Martino</surname><given-names>G</given-names> </name><name name-style="western"><surname>Degli Esposti</surname><given-names>M</given-names> </name></person-group><article-title>Overview of erisk 2022: early risk prediction on the internet</article-title><year>2022</year><conf-name>Experimental IR Meets Multilinguality, Multimodality, and Interaction (CLEF 2018)</conf-name><conf-date>Sep 5-8, 2022</conf-date><conf-loc>Bologna, Italy</conf-loc><fpage>233</fpage><lpage>256</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-13643-6_18</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Losada</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Crestani</surname><given-names>F</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Fuhr</surname><given-names>N</given-names> </name><name name-style="western"><surname>Quaresma</surname><given-names>P</given-names> </name><name name-style="western"><surname>Gon&#x00E7;alves</surname><given-names>T</given-names> </name><name name-style="western"><surname>Larsen</surname><given-names>B</given-names> </name><name name-style="western"><surname>Balog</surname><given-names>K</given-names> </name><name name-style="western"><surname>Macdonald</surname><given-names>C</given-names> </name><name name-style="western"><surname>Cappellato</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ferro</surname><given-names>N</given-names> </name></person-group><article-title>A test collection for research on depression and language use</article-title><conf-name>Experimental IR Meets Multilinguality, Multimodality, and Interaction: 7th International Conference of the CLEF Association (CLEF 2016)</conf-name><conf-date>Sep 5-8, 2016</conf-date><conf-loc>&#x00C9;vora, Portugal</conf-loc><fpage>28</fpage><lpage>39</lpage><pub-id pub-id-type="doi">10.1007/978-3-319-44564-9_3</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Losada</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Crestani</surname><given-names>F</given-names> </name><name name-style="western"><surname>Parapar</surname><given-names>J</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Arampatzis</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kanoulas</surname><given-names>E</given-names> </name><name name-style="western"><surname>Tsikrika</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Overview of erisk 2020: early risk prediction on the internet</article-title><source>Experimental IR Meets Multilinguality, Multimodality, and Interaction</source><year>2020</year><publisher-name>Springer International Publishing</publisher-name><fpage>272</fpage><lpage>287</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-58219-7_20</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Losada</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Crestani</surname><given-names>F</given-names> </name><name name-style="western"><surname>Parapar</surname><given-names>J</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Crestani</surname><given-names>F</given-names> </name><name name-style="western"><surname>Braschler</surname><given-names>M</given-names> </name><name name-style="western"><surname>Savoy</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Overview of erisk 2019 early risk prediction on the internet</article-title><source>Experimental IR Meets Multilinguality, Multimodality, and Interaction</source><year>2019</year><publisher-name>Springer International Publishing</publisher-name><fpage>340</fpage><lpage>357</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-28577-7_27</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Parapar</surname><given-names>J</given-names> </name><name name-style="western"><surname>Mart&#x00ED;n-Rodilla</surname><given-names>P</given-names> </name><name name-style="western"><surname>Losada</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Crestani</surname><given-names>F</given-names> </name></person-group><article-title>Overview of erisk at CLEF 2021: early risk prediction on the internet (extended overview)</article-title><source>Experimental IR Meets Multilinguality, Multimodality, and Interaction</source><year>2021</year><publisher-name>Springer International Publishing</publisher-name><fpage>324</fpage><lpage>344</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-85251-1_22</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="web"><article-title>DAIC-WOZ database &#x0026; extended DAIC database</article-title><source>University of Southern California</source><access-date>2025-10-21</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://dcapswoz.ict.usc.edu/">https://dcapswoz.ict.usc.edu/</ext-link></comment></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>S</given-names> </name><name name-style="western"><surname>Nair</surname><given-names>R</given-names> </name><name name-style="western"><surname>Naqvi</surname><given-names>SM</given-names> </name></person-group><article-title>Acoustic and text features analysis for adult ADHD screening: a data-driven approach utilizing DIVA interview</article-title><source>IEEE J Transl Eng Health Med</source><year>2024</year><volume>12</volume><fpage>359</fpage><lpage>370</lpage><pub-id pub-id-type="doi">10.1109/JTEHM.2024.3369764</pub-id><pub-id pub-id-type="medline">38606391</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Speier</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ong</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Arnold</surname><given-names>CW</given-names> </name></person-group><article-title>Bidirectional representation learning from transformers using multimodal electronic health record data to predict depression</article-title><source>IEEE J Biomed Health Inform</source><year>2021</year><month>08</month><volume>25</volume><issue>8</issue><fpage>3121</fpage><lpage>3129</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2021.3063721</pub-id><pub-id pub-id-type="medline">33661740</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Hayati</surname><given-names>MFM</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rosli</surname><given-names>A</given-names> </name></person-group><article-title>Depression detection on malay dialects using GPT-3</article-title><conf-name>2022 IEEE-EMBS Conference on Biomedical Engineering and Sciences (IECBES)</conf-name><conf-date>Dec 7-9, 2022</conf-date><conf-loc>Kuala Lumpur, Malaysia</conf-loc><fpage>360</fpage><lpage>364</lpage><pub-id pub-id-type="doi">10.1109/IECBES54088.2022.10079554</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>W</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Applying contrastive pre-training for depression and anxiety risk prediction in type 2 diabetes patients based on heterogeneous electronic health records: a primary healthcare case study</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>01</month><day>18</day><volume>31</volume><issue>2</issue><fpage>445</fpage><lpage>455</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad228</pub-id><pub-id pub-id-type="medline">38062850</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Levis</surname><given-names>B</given-names> </name><name name-style="western"><surname>Riehm</surname><given-names>KE</given-names> </name><etal/></person-group><article-title>Equivalency of the diagnostic accuracy of the PHQ-8 and PHQ-9: a systematic review and individual participant data meta-analysis</article-title><source>Psychol Med</source><year>2020</year><month>06</month><volume>50</volume><issue>8</issue><fpage>1368</fpage><lpage>1380</lpage><pub-id pub-id-type="doi">10.1017/S0033291719001314</pub-id><pub-id pub-id-type="medline">31298180</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Hamilton</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sartorius</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ban</surname><given-names>TA</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Sartorius</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ban</surname><given-names>TA</given-names> </name></person-group><article-title>The Hamilton Rating Scale for Depression</article-title><source>Assessment of Depression</source><year>1986</year><access-date>2025-10-24</access-date><publisher-name>Springer</publisher-name><fpage>143</fpage><lpage>152</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/978-3-642-70486-4_14">https://doi.org/10.1007/978-3-642-70486-4_14</ext-link></comment><pub-id pub-id-type="doi">10.1007/978-3-642-70486-4_14</pub-id></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Beck</surname><given-names>AT</given-names> </name><name name-style="western"><surname>Steer</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>G</given-names> </name></person-group><article-title>Beck Depression Inventory&#x2013;II</article-title><source>APA PsycNet</source><year>2011</year><access-date>2026-07-27</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://doi.apa.org/doi/10.1037/t00742-000">https://doi.apa.org/doi/10.1037/t00742-000</ext-link></comment></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Weathers</surname><given-names>FW</given-names> </name><name name-style="western"><surname>Litz</surname><given-names>B</given-names> </name><name name-style="western"><surname>Herman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Juska</surname><given-names>J</given-names> </name><name name-style="western"><surname>Keane</surname><given-names>T</given-names> </name></person-group><article-title>PTSD checklist&#x2014;civilian version</article-title><source>APA PsycNet</source><access-date>2026-04-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://psycnet.apa.org/doiLanding?doi=10.1037%2Ft02622-000">https://psycnet.apa.org/doiLanding?doi=10.1037%2Ft02622-000</ext-link></comment></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tombaugh</surname><given-names>TN</given-names> </name><name name-style="western"><surname>McIntyre</surname><given-names>NJ</given-names> </name></person-group><article-title>The Mini-Mental State Examination: a comprehensive review</article-title><source>J Am Geriatr Soc</source><year>1992</year><month>09</month><volume>40</volume><issue>9</issue><fpage>922</fpage><lpage>935</lpage><pub-id pub-id-type="doi">10.1111/j.1532-5415.1992.tb01992.x</pub-id><pub-id pub-id-type="medline">1512391</pub-id></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ohse</surname><given-names>J</given-names> </name><name name-style="western"><surname>Had&#x017E;i&#x0107;</surname><given-names>B</given-names> </name><name name-style="western"><surname>Mohammed</surname><given-names>P</given-names> </name><etal/></person-group><article-title>GPT-4 shows potential for identifying social anxiety from clinical interview data</article-title><source>Sci Rep</source><year>2024</year><month>12</month><day>16</day><volume>14</volume><issue>1</issue><fpage>30498</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-82192-2</pub-id><pub-id pub-id-type="medline">39681627</pub-id></nlm-citation></ref><ref id="ref81"><label>81</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ott</surname><given-names>M</given-names> </name><name name-style="western"><surname>Goyal</surname><given-names>N</given-names> </name><etal/></person-group><article-title>RoBERTa: a robustly optimized BERT pretraining approach</article-title><source>arXiv</source><access-date>2024-06-17</access-date><comment>Preprint posted online on  Jul 26, 2019</comment><comment><ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1907.11692">http://arxiv.org/abs/1907.11692</ext-link></comment><pub-id pub-id-type="doi">10.48550/arXiv.1907.11692</pub-id></nlm-citation></ref><ref id="ref82"><label>82</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Sanh</surname><given-names>V</given-names> </name><name name-style="western"><surname>Debut</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chaumond</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wolf</surname><given-names>T</given-names> </name></person-group><article-title>DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter</article-title><source>arXiv</source><access-date>2025-09-10</access-date><comment>Preprint posted online on  Mar 1, 2020</comment><comment><ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1910.01108">http://arxiv.org/abs/1910.01108</ext-link></comment><pub-id pub-id-type="doi">10.48550/arXiv.1910.01108</pub-id></nlm-citation></ref><ref id="ref83"><label>83</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Dai</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Carbonell</surname><given-names>J</given-names> </name><name name-style="western"><surname>Salakhutdinov</surname><given-names>R</given-names> </name><name name-style="western"><surname>Le</surname><given-names>QV</given-names> </name></person-group><article-title>XLNet: generalized autoregressive pretraining for language understanding</article-title><conf-name>Proceedings of the 33rd International Conference on Neural Information Processing Systems</conf-name><conf-date>Dec 8-14, 2019</conf-date><conf-loc>Red Hook, NY</conf-loc><fpage>5753</fpage><lpage>5763</lpage><pub-id pub-id-type="doi">10.5555/3454287.3454804</pub-id></nlm-citation></ref><ref id="ref84"><label>84</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ji</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ansari</surname><given-names>L</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tiwari</surname><given-names>P</given-names> </name><name name-style="western"><surname>Cambria</surname><given-names>E</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Calzolari</surname><given-names>N</given-names> </name><name name-style="western"><surname>B&#x00E9;chet</surname><given-names>F</given-names> </name><name name-style="western"><surname>Blache</surname><given-names>P</given-names> </name></person-group><article-title>MentalBERT: publicly available pretrained language models for mental healthcare</article-title><conf-name>Thirteenth Language Resources and Evaluation Conference</conf-name><conf-date>Jun 20-25, 2022</conf-date><conf-loc>Marseille, France</conf-loc><fpage>7184</fpage><lpage>7190</lpage><pub-id pub-id-type="doi">10.63317/2d7umxifj8wz</pub-id></nlm-citation></ref><ref id="ref85"><label>85</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Goodman</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gimpel</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>P</given-names> </name><name name-style="western"><surname>Soricut</surname><given-names>R</given-names> </name></person-group><article-title>ALBERT: a lite BERT for self-supervised learning of language representations</article-title><access-date>2025-09-10</access-date><comment>Preprint posted online on  Feb 9, 2020</comment><comment><ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1909.11942">http://arxiv.org/abs/1909.11942</ext-link></comment><pub-id pub-id-type="doi">10.48550/arXiv.1909.11942</pub-id></nlm-citation></ref><ref id="ref86"><label>86</label><nlm-citation citation-type="other"><person-group person-group-type="author"><collab>OpenAI</collab><name name-style="western"><surname>Achiam</surname><given-names>J</given-names> </name><name name-style="western"><surname>Adler</surname><given-names>S</given-names> </name><name name-style="western"><surname>Agarwalrwa</surname><given-names>S</given-names> </name><etal/></person-group><article-title>GPT-4 technical report</article-title><source>arXiv</source><access-date>2024-07-18</access-date><comment>Preprint posted online on  Mar 4, 2024</comment><comment><ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2303.08774">http://arxiv.org/abs/2303.08774</ext-link></comment></nlm-citation></ref><ref id="ref87"><label>87</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Conneau</surname><given-names>A</given-names> </name><name name-style="western"><surname>Khandelwal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Goyal</surname><given-names>N</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Jurafsky</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schluter</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tetreault</surname><given-names>J</given-names> </name></person-group><article-title>Unsupervised cross-lingual representation learning at scale [Webinar]</article-title><conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>Jul 5-10, 2020</conf-date><conf-loc>Online</conf-loc><fpage>8440</fpage><lpage>8451</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.747</pub-id></nlm-citation></ref><ref id="ref88"><label>88</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Reimers</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gurevych</surname><given-names>I</given-names> </name><name name-style="western"><surname>Inui</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>X</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Inui</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>X</given-names> </name></person-group><article-title>Sentence-BERT: sentence embeddings using siamese BERT-networks</article-title><conf-name>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</conf-name><conf-date>Nov 3-7, 2019</conf-date><conf-loc>Hong Kong, China</conf-loc><fpage>3982</fpage><lpage>3992</lpage><pub-id pub-id-type="doi">10.18653/v1/D19-1410</pub-id></nlm-citation></ref><ref id="ref89"><label>89</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Alsentzer</surname><given-names>E</given-names> </name><name name-style="western"><surname>Murphy</surname><given-names>J</given-names> </name><name name-style="western"><surname>Boag</surname><given-names>W</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Rumshisky</surname><given-names>A</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name></person-group><article-title>Publicly available clinical BERT embeddings</article-title><conf-name>Proceedings of the 2nd Clinical Natural Language Processing Workshop</conf-name><conf-date>Jun 7, 2019</conf-date><conf-loc>Minneapolis, MN</conf-loc><fpage>72</fpage><lpage>78</lpage><pub-id pub-id-type="doi">10.18653/v1/W19-1909</pub-id></nlm-citation></ref><ref id="ref90"><label>90</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>X</given-names> </name><name name-style="western"><surname>Qin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>TY</given-names> </name></person-group><article-title>MPNet: masked and permuted pre-training for language understanding</article-title><access-date>2026-07-30</access-date><conf-name>Proceedings of the 34th International Conference on Neural Information Processing Systems</conf-name><conf-date>Dec 6, 2020</conf-date><conf-loc>Red Hook, NY, United States</conf-loc><fpage>16857</fpage><lpage>16867</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/10.5555/3495724.3497138">https://dl.acm.org/doi/10.5555/3495724.3497138</ext-link></comment></nlm-citation></ref><ref id="ref91"><label>91</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Clark</surname><given-names>K</given-names> </name><name name-style="western"><surname>Luong</surname><given-names>MT</given-names> </name><name name-style="western"><surname>Le</surname><given-names>QV</given-names> </name><etal/></person-group><article-title>ELECTRA: pre-training text encoders as discriminators rather than generators</article-title><access-date>2026-07-30</access-date><conf-name>8th International Conference on Learning Representations</conf-name><conf-date>Apr 26-30, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://nlp.stanford.edu/pubs/clark2020electra.pdf">https://nlp.stanford.edu/pubs/clark2020electra.pdf</ext-link></comment></nlm-citation></ref><ref id="ref92"><label>92</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>P</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>W</given-names> </name></person-group><article-title>Deberta: decoding-enhanced bert with disentangled attention</article-title><access-date>2026-07-30</access-date><conf-name>9th International Conference on Learning Representations</conf-name><conf-date>May 3-7, 2021</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://iclr.cc/virtual/2021/poster/2562">https://iclr.cc/virtual/2021/poster/2562</ext-link></comment></nlm-citation></ref><ref id="ref93"><label>93</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Antoun</surname><given-names>W</given-names> </name><name name-style="western"><surname>Baly</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hajj</surname><given-names>H</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Al-Khalifa</surname><given-names>H</given-names> </name><name name-style="western"><surname>Magdy</surname><given-names>W</given-names> </name><name name-style="western"><surname>Darwish</surname><given-names>K</given-names> </name><name name-style="western"><surname>Elsayed</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mubarak</surname><given-names>H</given-names> </name></person-group><article-title>Transformer-based model for arabic language understanding</article-title><access-date>2026-07-30</access-date><conf-name>4th Workshop on Open-Source Arabic Corpora and Processing Tools, with a Shared Task on Offensive Language Detection</conf-name><conf-date>May 12, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.researchgate.net/publication/353371883_AraBERT_Transformer-based_Model_for_Arabic_Language_Understanding">https://www.researchgate.net/publication/353371883_AraBERT_Transformer-based_Model_for_Arabic_Language_Understanding</ext-link></comment></nlm-citation></ref><ref id="ref94"><label>94</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>W</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><etal/></person-group><article-title>BioBERT: a pre-trained biomedical language representation model for biomedical text mining</article-title><source>Bioinformatics</source><year>2020</year><month>02</month><day>15</day><volume>36</volume><issue>4</issue><fpage>1234</fpage><lpage>1240</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id><pub-id pub-id-type="medline">31501885</pub-id></nlm-citation></ref><ref id="ref95"><label>95</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Po&#x015B;wiata</surname><given-names>R</given-names> </name><name name-style="western"><surname>Pere&#x0142;kiewicz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chakravarthi</surname><given-names>BR</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Chakravarthi</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Bharathi</surname><given-names>B</given-names> </name><name name-style="western"><surname>McCrae</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Zarrouk</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bali</surname><given-names>K</given-names> </name><name name-style="western"><surname>Buitelaar</surname><given-names>P</given-names> </name></person-group><article-title>OPI@LT-EDI-ACL2022: detecting signs of depression from social media text using roberta pre-trained language models</article-title><conf-name>Proceedings of the Second Workshop on Language Technology for Equality, Diversity and Inclusion</conf-name><conf-date>May 27, 2022</conf-date><conf-loc>Dublin, Ireland</conf-loc><fpage>276</fpage><lpage>282</lpage><pub-id pub-id-type="doi">10.18653/v1/2022.ltedi-1.40</pub-id></nlm-citation></ref><ref id="ref96"><label>96</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>K</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name></person-group><article-title>Knowledge-enhanced pre-training large language model for depression diagnosis and treatment</article-title><conf-name>2023 IEEE 9th International Conference on Cloud Computing and Intelligent Systems (CCIS)</conf-name><conf-date>Aug 12-13, 2023</conf-date><conf-loc>Dali, China</conf-loc><fpage>532</fpage><lpage>536</lpage><pub-id pub-id-type="doi">10.1109/CCIS59572.2023.10263217</pub-id></nlm-citation></ref><ref id="ref97"><label>97</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>W</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name></person-group><article-title>Multimodal depression detection based on factorized representation [Webinar]</article-title><conf-name>2022 International Conference on High Performance Big Data and Intelligent Systems (HDIS)</conf-name><conf-date>Dec 10-11, 2022</conf-date><fpage>190</fpage><lpage>196</lpage><pub-id pub-id-type="doi">10.1109/HDIS56859.2022.9991717</pub-id></nlm-citation></ref><ref id="ref98"><label>98</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dhankar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Katz</surname><given-names>A</given-names> </name></person-group><article-title>Tracking pregnant women&#x2019;s mental health through social media: an analysis of Reddit posts</article-title><source>JAMIA Open</source><year>2023</year><month>12</month><volume>6</volume><issue>4</issue><fpage>ooad094</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooad094</pub-id><pub-id pub-id-type="medline">38033783</pub-id></nlm-citation></ref><ref id="ref99"><label>99</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Boonyarat</surname><given-names>P</given-names> </name><name name-style="western"><surname>Liew</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>YC</given-names> </name></person-group><article-title>Leveraging enhanced BERT models for detecting suicidal ideation in Thai social media content amidst COVID-19</article-title><source>Inf Process Manag</source><year>2024</year><month>07</month><volume>61</volume><issue>4</issue><fpage>103706</fpage><pub-id pub-id-type="doi">10.1016/j.ipm.2024.103706</pub-id></nlm-citation></ref><ref id="ref100"><label>100</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>EL</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>CY</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Chu</surname><given-names>KC</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>MS</given-names> </name></person-group><article-title>Development of internet suicide message identification and the Monitoring-Tracking-Rescuing model in Taiwan</article-title><source>J Affect Disord</source><year>2023</year><month>01</month><day>1</day><volume>320</volume><fpage>37</fpage><lpage>41</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2022.09.090</pub-id><pub-id pub-id-type="medline">36162682</pub-id></nlm-citation></ref><ref id="ref101"><label>101</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shimamoto</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ishizuka</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ohtani</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Machine learning algorithm-based estimation model for the severity of depression assessed using Montgomery-Asberg depression rating scale</article-title><source>Neuropsychopharmacol Rep</source><year>2024</year><month>03</month><volume>44</volume><issue>1</issue><fpage>115</fpage><lpage>120</lpage><pub-id pub-id-type="doi">10.1002/npr2.12404</pub-id><pub-id pub-id-type="medline">38115795</pub-id></nlm-citation></ref><ref id="ref102"><label>102</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Matero</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hung</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schwartz</surname><given-names>HA</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Barnes</surname><given-names>J</given-names> </name><name name-style="western"><surname>Clercq</surname><given-names>O</given-names> </name><name name-style="western"><surname>Barriere</surname><given-names>V</given-names> </name></person-group><article-title>Evaluating contextual embeddings and their extraction layers for depression assessment</article-title><conf-name>Proceedings of the 12th Workshop on Computational Approaches to Subjectivity, Sentiment &#x0026; Social Media Analysis</conf-name><conf-date>May 26, 2022</conf-date><conf-loc>Dublin, Ireland</conf-loc><fpage>89</fpage><lpage>94</lpage><pub-id pub-id-type="doi">10.18653/v1/2022.wassa-1.9</pub-id></nlm-citation></ref><ref id="ref103"><label>103</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khalil</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Tawfik</surname><given-names>NS</given-names> </name><name name-style="western"><surname>Spruit</surname><given-names>M</given-names> </name></person-group><article-title>Federated learning for privacy-preserving depression detection with multilingual language models in social media posts</article-title><source>Patterns (N Y)</source><year>2024</year><month>07</month><day>12</day><volume>5</volume><issue>7</issue><fpage>100990</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2024.100990</pub-id><pub-id pub-id-type="medline">39081573</pub-id></nlm-citation></ref><ref id="ref104"><label>104</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>McMahan</surname><given-names>B</given-names> </name><name name-style="western"><surname>Moore</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ramage</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hampson</surname><given-names>S</given-names> </name><name name-style="western"><surname>Arcas</surname><given-names>BY</given-names> </name></person-group><article-title>Communication-efficient learning of deep networks from decentralized data</article-title><access-date>2026-07-30</access-date><conf-name>Proceedings of the 20th International Conference on Artificial Intelligence and Statistics</conf-name><conf-date>Apr 20-22, 2017</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://research.google/pubs/communication-efficient-learning-of-deep-networks-from-decentralized-data/">https://research.google/pubs/communication-efficient-learning-of-deep-networks-from-decentralized-data/</ext-link></comment></nlm-citation></ref><ref id="ref105"><label>105</label><nlm-citation citation-type="web"><article-title>Mental disorders</article-title><source>World Health Organization</source><year>2025</year><access-date>2025-11-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/news-room/fact-sheets/detail/mental-disorders">https://www.who.int/news-room/fact-sheets/detail/mental-disorders</ext-link></comment></nlm-citation></ref><ref id="ref106"><label>106</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>B&#x00E5;&#x00E5;th</surname><given-names>R</given-names> </name><name name-style="western"><surname>Sikstr&#x00F6;m</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kalnak</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hansson</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sahl&#x00E9;n</surname><given-names>B</given-names> </name></person-group><article-title>Latent semantic analysis discriminates children with developmental language disorder (DLD) from children with typical language development</article-title><source>J Psycholinguist Res</source><year>2019</year><month>06</month><volume>48</volume><issue>3</issue><fpage>683</fpage><lpage>697</lpage><pub-id pub-id-type="doi">10.1007/s10936-018-09625-8</pub-id><pub-id pub-id-type="medline">30684119</pub-id></nlm-citation></ref><ref id="ref107"><label>107</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Bender</surname><given-names>EM</given-names> </name></person-group><article-title>The #BenderRule: on naming the languages we study and why it matters</article-title><source>The Gradient</source><year>2019</year><access-date>2025-10-24</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://thegradient.pub/the-benderrule-on-naming-the-languages-we-study-and-why-it-matters/">https://thegradient.pub/the-benderrule-on-naming-the-languages-we-study-and-why-it-matters/</ext-link></comment></nlm-citation></ref><ref id="ref108"><label>108</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ducel</surname><given-names>F</given-names> </name><name name-style="western"><surname>Fort</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lejeune</surname><given-names>G</given-names> </name><name name-style="western"><surname>Lepage</surname><given-names>Y</given-names> </name></person-group><article-title>Do we name the languages we study? the #benderrule in LREC and ACL articles</article-title><conf-name>Thirteenth Language Resources and Evaluation Conference</conf-name><conf-date>Jun 20-25, 2022</conf-date><conf-loc>Marseille, France</conf-loc><fpage>564</fpage><lpage>573</lpage><pub-id pub-id-type="doi">10.63317/3mqaq2qsx5mv</pub-id></nlm-citation></ref><ref id="ref109"><label>109</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sawhney</surname><given-names>R</given-names> </name><name name-style="western"><surname>Joshi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Gandhi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>RR</given-names> </name></person-group><article-title>Robust suicide risk assessment on social media via deep adversarial learning</article-title><source>J Am Med Inform Assoc</source><year>2021</year><month>07</month><day>14</day><volume>28</volume><issue>7</issue><fpage>1497</fpage><lpage>1506</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocab031</pub-id><pub-id pub-id-type="medline">33779728</pub-id></nlm-citation></ref><ref id="ref110"><label>110</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Benton</surname><given-names>A</given-names> </name><name name-style="western"><surname>Coppersmith</surname><given-names>G</given-names> </name><name name-style="western"><surname>Dredze</surname><given-names>M</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Hovy</surname><given-names>D</given-names> </name><name name-style="western"><surname>Spruit</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mitchell</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bender</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Strube</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wallach</surname><given-names>H</given-names> </name></person-group><article-title>Ethical research protocols for social media health research</article-title><conf-name>Proceedings of the First ACL Workshop on Ethics in Natural Language Processing</conf-name><conf-date>Apr 4, 2017</conf-date><conf-loc>Valencia, Spain</conf-loc><fpage>94</fpage><lpage>102</lpage><pub-id pub-id-type="doi">10.18653/v1/W17-1612</pub-id></nlm-citation></ref><ref id="ref111"><label>111</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vishwanath</surname><given-names>K</given-names> </name><name name-style="western"><surname>Alyakin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ghosh</surname><given-names>M</given-names> </name><etal/></person-group><article-title>General-purpose large language models outperform specialized clinical AI tools on medical benchmarks</article-title><source>Nat Med</source><year>2026</year><month>07</month><volume>32</volume><issue>7</issue><fpage>2405</fpage><lpage>2409</lpage><pub-id pub-id-type="doi">10.1038/s41591-026-04431-5</pub-id><pub-id pub-id-type="medline">42286322</pub-id></nlm-citation></ref><ref id="ref112"><label>112</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ahmed</surname><given-names>MI</given-names> </name><name name-style="western"><surname>Spooner</surname><given-names>B</given-names> </name><name name-style="western"><surname>Isherwood</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lane</surname><given-names>M</given-names> </name><name name-style="western"><surname>Orrock</surname><given-names>E</given-names> </name><name name-style="western"><surname>Dennison</surname><given-names>A</given-names> </name></person-group><article-title>A systematic review of the barriers to the implementation of artificial intelligence in healthcare</article-title><source>Cureus</source><year>2023</year><month>10</month><volume>15</volume><issue>10</issue><fpage>e46454</fpage><pub-id pub-id-type="doi">10.7759/cureus.46454</pub-id><pub-id pub-id-type="medline">37927664</pub-id></nlm-citation></ref><ref id="ref113"><label>113</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Morand</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ligozat</surname><given-names>AL</given-names> </name><name name-style="western"><surname>N&#x00E9;v&#x00E9;ol</surname><given-names>A</given-names> </name></person-group><article-title>MLCA: a tool for machine learning life cycle assessment</article-title><conf-name>2024 10th International Conference on ICT for Sustainability (ICT4S)</conf-name><conf-date>Jun 24-28, 2024</conf-date><conf-loc>Stockholm, Sweden</conf-loc><fpage>227</fpage><lpage>238</lpage><pub-id pub-id-type="doi">10.1109/ICT4S64576.2024.00031</pub-id></nlm-citation></ref><ref id="ref114"><label>114</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Forsythe</surname><given-names>DE</given-names> </name></person-group><article-title>Engineering knowledge: the construction of knowledge in artificial intelligence</article-title><source>Soc Stud Sci</source><year>1993</year><month>08</month><volume>23</volume><issue>3</issue><fpage>445</fpage><lpage>477</lpage><pub-id pub-id-type="doi">10.1177/0306312793023003002</pub-id></nlm-citation></ref><ref id="ref115"><label>115</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Littmann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Selig</surname><given-names>K</given-names> </name><name name-style="western"><surname>Cohen-Lavi</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Validity of machine learning in biology and medicine increased through collaborations across fields of expertise</article-title><source>Nat Mach Intell</source><year>2020</year><month>01</month><volume>2</volume><issue>1</issue><fpage>18</fpage><lpage>24</lpage><pub-id pub-id-type="doi">10.1038/s42256-019-0139-8</pub-id></nlm-citation></ref><ref id="ref116"><label>116</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Esackimuthu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hariprasad</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sivanaiah</surname><given-names>R</given-names> </name><name name-style="western"><surname>S</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rajendram</surname><given-names>SM</given-names> </name><name name-style="western"><surname>T T</surname><given-names>M</given-names> </name></person-group><article-title>SSN_MLRG3 @lt-edi-acl2022-depression detection system from social media text using transformer models</article-title><conf-name>Proceedings of the Second Workshop on Language Technology for Equality, Diversity and Inclusion</conf-name><conf-date>May 27, 2022</conf-date><pub-id pub-id-type="doi">10.18653/v1/2022.ltedi-1.26</pub-id></nlm-citation></ref><ref id="ref117"><label>117</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Assessing depression risk in chinese microblogs: a corpus and machine learning methods</article-title><conf-name>2019 IEEE International Conference on Healthcare Informatics (ICHI)</conf-name><conf-date>Jun 10-13, 2019</conf-date><conf-loc>Xi&#x2019;an, China</conf-loc><fpage>1</fpage><lpage>5</lpage><pub-id pub-id-type="doi">10.1109/ICHI.2019.8904506</pub-id><pub-id pub-id-type="medline">32537571</pub-id></nlm-citation></ref><ref id="ref118"><label>118</label><nlm-citation citation-type="web"><article-title>Depression MeSH descriptor data 2026</article-title><source>National Library of Medicine</source><access-date>2025-10-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://meshb.nlm.nih.gov/record/ui?ui=D003863">https://meshb.nlm.nih.gov/record/ui?ui=D003863</ext-link></comment></nlm-citation></ref><ref id="ref119"><label>119</label><nlm-citation citation-type="web"><article-title>Major depressive disorder MeSH descriptor data 2026</article-title><source>National Library of Medicine</source><year>2025</year><access-date>2025-10-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://meshb.nlm.nih.gov/record/ui?ui=D003865">https://meshb.nlm.nih.gov/record/ui?ui=D003865</ext-link></comment></nlm-citation></ref><ref id="ref120"><label>120</label><nlm-citation citation-type="web"><article-title>Depressive disorder MeSH descriptor data 2026</article-title><source>National Library of Medicine</source><access-date>2025-10-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://meshb.nlm.nih.gov/record/ui?ui=D003866">https://meshb.nlm.nih.gov/record/ui?ui=D003866</ext-link></comment></nlm-citation></ref><ref id="ref121"><label>121</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goldsack</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Coravos</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bakker</surname><given-names>JP</given-names> </name><etal/></person-group><article-title>Verification, analytical validation, and clinical validation (V3): the foundation of determining fit-for-purpose for Biometric Monitoring Technologies (BioMeTs)</article-title><source>NPJ Digit Med</source><year>2020</year><volume>3</volume><fpage>55</fpage><pub-id pub-id-type="doi">10.1038/s41746-020-0260-4</pub-id><pub-id pub-id-type="medline">32337371</pub-id></nlm-citation></ref><ref id="ref122"><label>122</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bakker</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Barge</surname><given-names>R</given-names> </name><name name-style="western"><surname>Centra</surname><given-names>J</given-names> </name><etal/></person-group><article-title>V3+ extends the V3 framework to ensure user-centricity and scalability of sensor-based digital health technologies</article-title><source>NPJ Digit Med</source><year>2025</year><month>01</month><day>24</day><volume>8</volume><issue>1</issue><fpage>51</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01322-2</pub-id><pub-id pub-id-type="medline">39856145</pub-id></nlm-citation></ref><ref id="ref123"><label>123</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wornow</surname><given-names>M</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Thapa</surname><given-names>R</given-names> </name><etal/></person-group><article-title>The shaky foundations of large language models and foundation models for electronic health records</article-title><source>NPJ Digit Med</source><year>2023</year><month>07</month><day>29</day><volume>6</volume><issue>1</issue><fpage>135</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00879-8</pub-id><pub-id pub-id-type="medline">37516790</pub-id></nlm-citation></ref><ref id="ref124"><label>124</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grant</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Booth</surname><given-names>A</given-names> </name></person-group><article-title>A typology of reviews: an analysis of 14 review types and associated methodologies</article-title><source>Health Info Libraries J</source><year>2009</year><month>06</month><volume>26</volume><issue>2</issue><fpage>91</fpage><lpage>108</lpage><pub-id pub-id-type="doi">10.1111/j.1471-1842.2009.00848.x</pub-id></nlm-citation></ref><ref id="ref125"><label>125</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wagstaff</surname><given-names>KL</given-names> </name></person-group><article-title>Machine learning that matters</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 18, 2012</comment><pub-id pub-id-type="doi">10.48550/arXiv.1206.4656</pub-id></nlm-citation></ref><ref id="ref126"><label>126</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Raji</surname><given-names>ID</given-names> </name><name name-style="western"><surname>Bender</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Paullada</surname><given-names>A</given-names> </name><name name-style="western"><surname>Denton</surname><given-names>E</given-names> </name><name name-style="western"><surname>Hanna</surname><given-names>A</given-names> </name></person-group><article-title>AI and the everything in the whole wide world benchmark</article-title><access-date>2026-07-30</access-date><conf-name>35th Conference on Neural Information Processing Systems (NeurIPS 2021)</conf-name><conf-date>Dec 6-14, 2021</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://datasets-benchmarks-proceedings.neurips.cc/paper/2021/hash/084b6fbb10729ed4da8c3d3f5a3ae7c9-Abstract-round2.html">https://datasets-benchmarks-proceedings.neurips.cc/paper/2021/hash/084b6fbb10729ed4da8c3d3f5a3ae7c9-Abstract-round2.html</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Search strategy for studies inclusion with full string requests.</p><media xlink:href="ai_v5i1e88082_app1.docx" xlink:title="DOCX File, 10 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Extraction guide detailing the entities extracted, the rationale for inclusion, and operationalization.</p><media xlink:href="ai_v5i1e88082_app2.pdf" xlink:title="PDF File, 75 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Table of extracted and analyzed entities from the reviewed papers.</p><media xlink:href="ai_v5i1e88082_app3.xlsx" xlink:title="XLSX File, 46 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Complete list of included studies with metadata.</p><media xlink:href="ai_v5i1e88082_app4.xlsx" xlink:title="XLSX File, 27 KB"/></supplementary-material><supplementary-material id="app5"><label>Checklist 1</label><p>PRISMA-ScR checklist.</p><media xlink:href="ai_v5i1e88082_app5.pdf" xlink:title="PDF File, 36174 KB"/></supplementary-material></app-group></back></article>