<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e95063</article-id><article-id pub-id-type="doi">10.2196/95063</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Automated Fidelity Monitoring of Lay-Delivered Mental Health Interventions Using Large Language Models: Development and Pilot Validation of shamiriAI in Kenya</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Lilan</surname><given-names>Shadrack</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mochama</surname><given-names>Brandon</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Osborn</surname><given-names>Tom</given-names></name><degrees>BA</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mmbone</surname><given-names>Wendy</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kilonzo</surname><given-names>Rachael</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kamau</surname><given-names>Faith</given-names></name><degrees>BA</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Daya</surname><given-names>Rahim</given-names></name><degrees>MA</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wasanga</surname><given-names>Christine</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Shamiri Institute</institution><addr-line>13th Floor, CMS Africa, Chania Avenue</addr-line><addr-line>Nairobi</addr-line><country>Kenya</country></aff><aff id="aff2"><institution>Department of Psychology, Kenyatta University</institution><addr-line>Nairobi</addr-line><addr-line>Nairobi County</addr-line><country>Kenya</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Rognli</surname><given-names>Erling</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Shadrack Lilan, BSc, Shamiri Institute, 13th Floor, CMS Africa, Chania Avenue, Nairobi, 00505, Kenya, +254 (0) 112540760; <email>shadrack.lilan@shamiri.institute</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>23</day><month>7</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e95063</elocation-id><history><date date-type="received"><day>10</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>12</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>15</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Shadrack Lilan, Brandon Mochama, Tom Osborn, Wendy Mmbone, Rachael Kilonzo, Faith Kamau, Rahim Daya, Christine Wasanga. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 23.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e95063"/><abstract><sec><title>Background</title><p>Task-shifting can help close the mental health treatment gap in low- and middle-income countries, but its effectiveness depends on ongoing supervision, which is hard to scale. AI tools that process session recordings and generate structured fidelity feedback could offer a scalable alternative; yet, to our knowledge, none have been developed or validated for lay-delivered, multilingual, group-format interventions in low-resource settings.</p></sec><sec><title>Objective</title><p>We developed and pilot-validated shamiriAI (ShamiriAI Institute), an automated fidelity-monitoring tool for lay-delivered mental health interventions, embedded within the Shamiri school-based program in Kenya.</p></sec><sec sec-type="methods"><title>Methods</title><p>Across 6 secondary schools in Ngong Hub, Kajiado County, Kenya (May-September 2025), shamiriAI processed session audio from 47 lay providers through a 5-stage pipeline: ingestion, multilingual automatic speech recognition (ASR) with prosodic feature extraction, personally identifiable information scrubbing, large language model&#x2013;based fidelity inference, and supervisor reports. The following two aims were assessed: (1) ASR performance on a held-out test set of manually transcribed sessions and (2) interrater reliability between shamiriAI and independent human supervisor ratings across 52 sessions (38 AI-augmented and 14 standard) on 6 domains (Required Contents, Specifics, Thoroughness, Clarity, Skill, and Purity; 1&#x2010;7 scale). Reliability used intraclass correlation coefficients, Bland-Altman analysis, adjacent-agreement rates, paired <italic>t</italic> tests with Holm-Bonferroni correction, and Gwet AC2 (ordinal weights) across 3 formulations of the human reference.</p></sec><sec sec-type="results"><title>Results</title><p>The ASR model achieved a character error rate of 0.19, word error rate of 0.34, and cosine semantic similarity of 0.77, indicating strong meaning preservation in code-switched speech. AI fidelity scores were systematically lower than the human composite overall (mean 5.14, SD 0.77 vs mean 5.93, SD 0.57 ); &#x0394;=&#x2212;0.79; d=&#x2212;1.16; <italic>P</italic>&#x003C;.001). Primary intraclass correlation coefficients ranged from &#x2212;0.06 to 0.20 across the 6 domains, and AC2 sensitivity analyses (against each individual rater and the rounded composite) corroborated this dimension-level ordering. Three patterns emerged: large systematic underrating on holistic dimensions (Required Contents: d=&#x2212;3.48; Clarity d=&#x2212;1.56); bidirectional medium-effect bias on facilitation dimensions (Thoroughness d=&#x2212;0.99; Skill d=+0.87); and substantial agreement on Specifics, Skill, and Purity (Gwet AC2 0.69&#x2010;0.76 against the rounded composite), relative to a human-human AC2 ceiling of 0.42&#x2010;0.60 estimated in the same dataset. Exploratory, underpowered subgroup and per-arm checks found no preliminary evidence of bias by lay-provider sex or age band or of arm-level differences.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Within this 52-session pilot, shamiriAI shows technically feasible multilingual ASR and a coherent, dimension-dependent reliability profile. Specifics, Purity, and Skill already reach substantial agreement, while underperformance on holistic dimensions (Required Contents and Clarity) reflects diagnosable misalignments in rubric interpretation and prompt design, specifying a concrete agenda for shamiriAI (version 2; Shamiri Institute). Whether AI-augmented supervision improves provider skill or student mental health outcomes will be tested in a planned cluster-randomized noninferiority trial.</p></sec><sec><title>Trial Registration</title><p>Pan African Clinical Trials Registry PACTR202508900479778; <ext-link ext-link-type="uri" xlink:href="https://pactr.samrc.ac.za/TrialDisplay.aspx?TrialID=33575">https://pactr.samrc.ac.za/TrialDisplay.aspx?TrialID=33575</ext-link></p></sec></abstract><kwd-group><kwd>fidelity monitoring</kwd><kwd>task-shifting</kwd><kwd>automatic speech recognition</kwd><kwd>large language models</kwd><kwd>clinical supervision</kwd><kwd>adolescent mental health</kwd><kwd>multilingual</kwd><kwd>low- and middle-income countries</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>We are in the middle of a global mental health crisis among young people [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Depression, anxiety, and related disorders are among the leading causes of illness and disability in people aged 10 to 24 [<xref ref-type="bibr" rid="ref4">4</xref>], and their prevalence has risen sharply over the past 3 decades [<xref ref-type="bibr" rid="ref2">2</xref>]. An estimated 279 million young people aged 10-24 years live with a mental health disorder [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. Most receive no help. The treatment gap&#x2014;the proportion of people who need care but do not receive it&#x2014;is 40%-60% globally and exceeds 80% in low-resource settings, where the majority of the world&#x2019;s young people live [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>The core structural problem driving this gap is a global scarcity of providers [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. The global median is 13.5 specialized mental health workers per 100,000 people [<xref ref-type="bibr" rid="ref8">8</xref>]; the ratio falls to just 1.1&#x2010;2.4 per 100,000 in low- and middle-income countries, against 67.2 in high-income countries (HICs) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Kenya, for instance, where 1 in 3 adolescents experience a mental health problem [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref13">13</xref>], has about 2 specialists per 100,000 people, most of them concentrated in urban centers [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. Closing the treatment gap through specialist-led care alone is not feasible in the near term [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>Task-shifting&#x2014;training non-specialists to deliver structured, evidence-based interventions under supervision&#x2014;has emerged as the leading strategy for expanding the workforce and closing the treatment gap [<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref19">19</xref>]. Randomized trials across sub-Saharan Africa and South Asia demonstrate that lay providers&#x2014;including community health workers [<xref ref-type="bibr" rid="ref20">20</xref>], trained youth [<xref ref-type="bibr" rid="ref21">21</xref>], and even grandmothers [<xref ref-type="bibr" rid="ref22">22</xref>]&#x2014;can deliver effective treatments for depression, anxiety, and trauma, producing clinically meaningful improvements that are sometimes comparable to those achieved by professionals [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref23">23</xref>].</p><p>The Shamiri Model in Kenya illustrates what task-shifting looks like in practice. Shamiri is a brief, 4-week group intervention for secondary school students built around strengths-based psychological principles&#x2014;growth mindset, gratitude, and values affirmation [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref24">24</xref>]. It operates through a 3-tier delivery model. In the first tier, lay providers aged 18-22 years are trained to deliver weekly group sessions to cohorts of 6 to 15 students [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. Clinical supervisors with backgrounds in psychology or social work form the second tier, overseeing lay providers and providing direct one-on-one support where needed [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. Complex or severe cases are escalated to licensed clinicians in the third tier [<xref ref-type="bibr" rid="ref26">26</xref>]. The structure maximizes reach by placing trained lay providers at the point of care; the model operates at the school level&#x2014;where adolescents spend most of their time&#x2014;without drawing on scarce specialists for routine cases [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]. Randomized controlled trials have demonstrated that the model significantly reduces depression and anxiety symptoms in secondary school students, with effects sustained at least 7 months postintervention [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>].</p><p>Task-shifting has demonstrated that an expanded workforce can deliver effective care [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. The harder problem is maintaining quality across that workforce as programs grow. Training a cohort of 20 lay providers is manageable; supervising hundreds of them, across dozens of sites, delivering sessions week after week, is a fundamentally different challenge&#x2014;and one that existing models have not solved [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>].</p><p>The answer, in every evidence-based framework for task-shifting, is supervision [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Supervision is the mechanism through which lay providers consolidate skills, receive corrective feedback, maintain fidelity to protocol, manage clinical risk, and sustain the psychological safety required to work with vulnerable youth [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref32">32</xref>-<xref ref-type="bibr" rid="ref34">34</xref>]. Without it, skill drift is invisible&#x2014;sessions degrade in quality, required elements go undelivered, and program managers have no way of knowing [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>].</p><p>But high-quality supervision is difficult to scale. Traditional supervision models require trained professionals to manually review session recordings, meet individually or in groups with providers, and deliver timely, specific feedback [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. As programs grow, supervision becomes a bottleneck. Supervisors can review only a fraction of sessions, feedback arrives late, and important behavioral information&#x2014;tone, pacing, and interpersonal dynamics&#x2014;is lost [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]. Supervision remains a central constraint on the quality and scalability of task-shifted mental health care.</p><p>Two advances offer a route forward. The first is the maturation of task-shifting models themselves. Models like Shamiri have shown that lay-delivered interventions can be implemented at scale with structured training and supervision systems&#x2014;creating the operational infrastructure into which quality-assurance tools can be embedded [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. The second is the emergence of AI tools capable of automating the observational and feedback functions of supervision. Advances in automatic speech recognition (ASR), natural language processing, and large language models (LLMs) have made it technically feasible to process session audio, estimate provider behaviors, score fidelity to clinical protocols, and generate structured feedback reports&#x2014;potentially reducing the manual-review bottleneck through systematic, scalable monitoring [<xref ref-type="bibr" rid="ref36">36</xref>-<xref ref-type="bibr" rid="ref39">39</xref>]. Research teams in HICs have developed AI systems that achieve human-level reliability on session-quality ratings for individual cognitive behavioral therapy (CBT) in English [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref39">39</xref>], and recent work demonstrates that LLMs can score therapeutic constructs from transcribed sessions with strong psychometric properties and valid associations with outcomes [<xref ref-type="bibr" rid="ref40">40</xref>]. The trajectory suggests AI-assisted fidelity monitoring is becoming viable&#x2014;but viable under a specific set of conditions: individual therapy, adult populations, professional providers, and English-language settings in HICs.</p><p>Whether AI fidelity monitoring can be extended to the conditions that define most of the world&#x2019;s task-shifted mental health care is entirely unestablished. Group delivery, lay providers, multilingual code-switched speech, minimal training data, and resource-constrained deployment contexts are not peripheral variations&#x2014;they are the structural features of the settings where supervision is most needed and least available. To our knowledge, no automated fidelity-monitoring system has been developed for, or validated in, this combination of lay-delivered, multilingual, group-format care. The gap spans every axis that matters, including delivery format, workforce type, linguistic environment, and implementation constraints.</p></sec><sec id="s1-2"><title>Objectives</title><p>To address these gaps, we developed shamiriAI&#x2014;an AI system designed to support fidelity monitoring and supervision for lay-delivered mental health interventions in multilingual, low-resource settings. shamiriAI processes raw session audio, performs multilingual ASR across English, Kiswahili, and Sheng, extracts prosodic features capturing group dynamics and facilitator behavior, and generates structured written feedback reports for clinical supervisors. It operates in an AI-in-the-loop configuration: supervisors remain the primary agents of support and clinical judgment; AI-generated feedback extends their observational reach, providing structured input on sessions that would otherwise go unreviewed.</p><p>This study reports the foundational validation work required before AI-augmented supervision can be meaningfully tested for its effect on provider skill or student outcomes. Before asking whether the system works clinically, 2 prior questions must be answered: can it accurately transcribe sessions in this linguistic environment, and are its fidelity assessments sufficiently aligned with expert human judgment to constitute credible supervisory inputs?</p><p>The objectives of this pilot study were twofold. The first was to evaluate the performance of shamiriAI&#x2019;s multilingual ASR pipeline on a held-out test set of manually transcribed sessions, using metrics appropriate for code-switched, agglutinative speech. The second was to assess interrater reliability between shamiriAI-generated fidelity ratings and independent human supervisor ratings across 6 fidelity dimensions, examining the degree of agreement, the direction and magnitude of systematic bias, and the specific dimensions for which the AI most and least closely approximated expert human judgment.</p><p>These aims are explicitly foundational. They establish whether shamiriAI produces transcripts and fidelity ratings with sufficient accuracy to support future development&#x2014;the essential first step toward an AI quality assurance infrastructure capable of supporting the supervision of lay providers at population scale across sub-Saharan Africa.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Setting</title><p>We conducted a pilot validation study of shamiriAI, an AI-based supervision support tool, embedded within routine delivery of the Shamiri intervention across 6 secondary schools in Ngong Hub, Kajiado County, Kenya, between May and September 2025. The study evaluated the following two pilot aims: (1) the technical performance of shamiriAI&#x2019;s ASR pipeline on a held-out test set of manually transcribed sessions, and (2) interrater reliability between shamiriAI-generated fidelity ratings and independent human supervisor ratings across 52 recorded sessions. These aims concern the technical properties of the system, not between-arm comparisons.</p><p>Lay providers (Fellows) were randomized to supervision arms at the start of the study. The sessions later recorded for validation were sampled independently at each site by lottery, without stratification by arm. Because assignment preceded session sampling, the validation set is imbalanced across arms (38 AI-augmented and 14 standard); the sampling mechanism is detailed under the Participants subheading.</p><p>The pilot was part of a larger 5-part implementation study examining optimization strategies for the Shamiri Model [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. The complete protocol is available in Supplement A in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-2"><title>Participants</title><p>Lay providers (Shamiri Fellows) are Kenyan young adults aged 18&#x2010;24 years recruited and trained to deliver the Shamiri intervention (refer to the Procedures subheading). All 64 Shamiri Fellows assigned to Ngong Hub were eligible; sex and age were collected via self-report. The 47 Fellows who participated in the parallel A/B test were randomized at project start to AI-augmented (n=34) or standard (n=13) supervision.</p><p>Audio recording for fidelity validation was a separate procedure. At each school session, the Hub Coordinator wrote the names of all lay providers present on slips of paper and drew 3 without replacement for recording, with no stratification by arm. This produced 52 recorded sessions, distributed unevenly across arms (38 AI-augmented and 14 standard) and across providers, reflecting natural variation in attendance and recording opportunity rather than a prespecified arm-balanced design.</p><p>Lay providers led group sessions for students aged 12&#x2010;21 years participating in the routine Shamiri intervention ([<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]; <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement A). Students were assigned to groups using a ballot-box procedure.</p></sec><sec id="s2-3"><title>About the Shamiri Intervention</title><p>Shamiri is a brief, school-based group intervention targeting adolescent depression and anxiety [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. Its components derive from the science of strengths-based interventions (overlapping with &#x201C;wise&#x201D; interventions [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]): brief, simple techniques targeting specific psychological processes [<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref46">46</xref>]&#x2014;growth mindset [<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>], gratitude [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>], and values affirmation [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. Complete protocols are published elsewhere [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref26">26</xref>]; refer to <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement A.</p></sec><sec id="s2-4"><title>Procedures</title><sec id="s2-4-1"><title>Lay Provider Recruitment, Training, and Supervision</title><sec id="s2-4-1-1"><title>Eligibility and Recruitment</title><p>Lay providers met the following four criteria: (1) at least aged 18 years; (2) completed secondary school in Kenya with English as the language of instruction; (3) able to read intervention protocols in English; and (4) available for all scheduled sessions [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. We recruited participants openly through WhatsApp (Meta) groups, university forums, and online job boards. Candidates completed online applications and structured 30-minute interviews assessing interest, relevant experience, personal characteristics conducive to group leadership, and responses to hypothetical implementation scenarios [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. Of the applicant pool, 64 were assigned to the Ngong Hub and were eligible for this study.</p></sec><sec id="s2-4-1-2"><title>Training</title><p>Training followed the standard Shamiri protocol [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref27">27</xref>] over 2 days and covering counseling techniques, group leadership, emergency and risk management, intervention didactics, and extensive role-playing, including approximately 6 hours of practice leading sessions while receiving real-time feedback from supervisors and peers [<xref ref-type="bibr" rid="ref25">25</xref>]. All lay providers were evaluated using a standardized rubric assessing content accuracy, timing, and subjective quality. Those who scored below standard received additional practice and follow-up assessment before leading groups. The protocol has been published elsewhere [<xref ref-type="bibr" rid="ref25">25</xref>] and has been used across multiple Shamiri trials and implementations [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>].</p></sec><sec id="s2-4-1-3"><title>Supervision</title><p>Each lay provider was assigned to a clinical supervisor and attended weekly 1-hour supervision sessions structured around reflection, feedback, and peer learning [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>] (refer to <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement B).</p></sec></sec></sec><sec id="s2-5"><title>Clinical Supervisor Recruitment, Training, and Supervision</title><p>Supervisors recruited, trained, and supervised lay providers while ensuring fidelity, managed clinically elevated cases, and triaged emergencies to the Clinical Network of expert psychiatrists and psychologists [<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref27">27</xref>]. All held at least a bachelor&#x2019;s degree in clinical or counseling psychology and were registered with the Kenya Counselors and Psychologists Board. Each supervised approximately 10 lay providers (1:10). Recruitment used a written interview and 2 case interviews with the Shamiri Clinical Team [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Supervisors then completed approximately 3 months of training covering the intervention manual and session-by-session protocols; clinical supervision techniques, including structured feedback delivery, fidelity rating using the 6-domain instrument, and calibration exercises; group facilitation and peer-counseling skills; identification and triage of elevated cases through the Clinical Network; ethical conduct and professional boundaries; and emergency risk management via the validated Shamiri Risk Management Protocol [<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref27">27</xref>]. The Shamiri Risk Management Protocol uses a 3-tier model (lay provider, caseworker or supervisor, and clinical expert) with standardized procedures for recognizing distress, initiating referrals, managing sensitive disclosures, and maintaining boundaries [<xref ref-type="bibr" rid="ref26">26</xref>]. Supervisors were also trained in data entry and quality control using internal software [<xref ref-type="bibr" rid="ref53">53</xref>].</p></sec><sec id="s2-6"><title>Study Conditions</title><sec id="s2-6-1"><title>Standard Supervision</title><p>All lay providers received weekly supervision from clinical supervisors [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Supervisors held weekly one-to-one or small-group sessions (~60 min) reviewing experiences and challenges, discussing selected sessions based on self-report and available notes, reinforcing adherence to the manual, and coaching on group management, risk management, and self-care [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Supervisors occasionally requested audio recordings, but systematic listening to and rating of entire sessions was not feasible given time constraints.</p></sec><sec id="s2-6-2"><title>AI-Augmented Supervision (ShamiriAI)</title><p>Supervisors continued all standard activities but additionally received structured shamiriAI feedback reports for recorded sessions, which they reviewed before or during supervision to guide discussions. Lay providers were not directly exposed to the AI system; all AI-generated feedback was delivered to supervisors, who relayed it during supervision.</p></sec></sec><sec id="s2-7"><title>Audio Recording Procedures</title><p>Lay providers recorded group sessions using study-supplied digital voice recorders, which captured wide-angle audio in open or semiopen spaces (eg, classrooms, halls, and outdoor areas), producing variable background noise. Recording coverage varied across sessions due to device failures, environmental noise, and school-specific restrictions. We included recordings from all 52 delivered sessions for which a recording was available (100% of recorded sessions). In the AI-augmented arm, Hub Coordinators uploaded files to secure cloud storage after each session; in the standard arm, recordings were archived for potential later analysis but were not processed by shamiriAI during the trial.</p></sec><sec id="s2-8"><title>Development of shamiriAI</title><sec id="s2-8-1"><title>Overview</title><p>shamiriAI enhances fidelity monitoring and supervision of lay-delivered group interventions. It processes raw session audio and produces structured feedback reports for clinical supervisors, covering both content- and process-related aspects of delivery. It operates in a code-switched multilingual environment (English, Kiswahili, and Sheng) characteristic of Kenyan secondary schools. It is a clinician-facing decision-support tool in which supervisors remain the primary decision-makers, and AI-generated feedback is one structured input alongside their direct clinical knowledge.</p><p>shamiriAI follows a linear 5-stage pipeline: (1) ingestion and preprocessing, (2) core processing (ASR and prosodic feature extraction, performed in parallel), (3) postprocessing and personally identifiable information (PII) scrubbing, (4) LLM-based feedback inference, and (5) delivery. <xref ref-type="table" rid="table1">Table 1</xref> summarizes each stage. All processing occurred on secure servers with restricted access.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Overview of the 5-stage shamiriAI pipeline. Stages 2a and 2b run in parallel on preprocessed audio. The pipeline ingests raw session audio and produces a structured fidelity feedback report for clinical supervisors.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Stage</td><td align="left" valign="bottom">Inputs</td><td align="left" valign="bottom">Outputs</td><td align="left" valign="bottom">Main processes</td><td align="left" valign="bottom">Key tools</td></tr></thead><tbody><tr><td align="left" valign="top">1. Ingestion and preprocessing</td><td align="left" valign="top">Session audio (.wav, .mp3)</td><td align="left" valign="top">Preprocessed audio segments</td><td align="left" valign="top">Format normalization (16 kHz, mono); denoising and spectral gating; VAD<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top">librosa, pydub, webrtcvad</td></tr><tr><td align="left" valign="top">2a. Automatic speech recognition</td><td align="left" valign="top">Preprocessed audio segments</td><td align="left" valign="top">Timestamped transcripts with speaker IDs</td><td align="left" valign="top">Fine-tuned Whisper (small, ~242M params); multilingual automatic speech recognition (English-Kiswahili-Sheng); speaker diarization (local labels); hallucination detection via VAD metadata</td><td align="left" valign="top">Whisper (fine-tuned), pyannote.audio</td></tr><tr><td align="left" valign="top">2b. Prosodic feature extraction</td><td align="left" valign="top">Preprocessed audio segments</td><td align="left" valign="top">Session- and segment-level feature vectors</td><td align="left" valign="top">Spectral features (centroid, bandwidth, contrast); speech-segment analysis (turns, gaps, and exchange ratio); voice diversity (pitch, energy variation); temporal dynamics and voice-quality indicators</td><td align="left" valign="top">librosa, parselmouth, python</td></tr><tr><td align="left" valign="top">3. Postprocessing and PII<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> scrubbing</td><td align="left" valign="top">Raw transcripts from 2a</td><td align="left" valign="top">Privacy-protected transcripts</td><td align="left" valign="top">NER<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> for PII<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> detection; masking of names, schools, locations; secure storage of unredacted audio or transcripts</td><td align="left" valign="top">NuNerZero (numind/NuNerZero); dslim/bert-base-NER<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> (fallback)</td></tr><tr><td align="left" valign="top">4. LLM<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup>-based feedback inference</td><td align="left" valign="top">Redacted transcripts + prosodic features</td><td align="left" valign="top">Structured feedback reports</td><td align="left" valign="top">Structured system prompt with intervention context; integration of transcript and prosodic summaries; fidelity assessment across 6 domains; iterative prompt refinement via clinical review</td><td align="left" valign="top">Gemini 2.5 Pro; custom prompt templates</td></tr><tr><td align="left" valign="top">5. Feedback delivery</td><td align="left" valign="top">Structured feedback from Stage 4</td><td align="left" valign="top">Supervisor-facing PDF reports</td><td align="left" valign="top">PDF generation with session metadata; quantitative prosodic summaries; narrative feedback (strengths, improvements, and suggestions); secure delivery via email/shared folder</td><td align="left" valign="top">ReportLab; secure cloud storage</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>VAD: voice activity detection.</p></fn><fn id="table1fn2"><p><sup>b</sup>PII: personally identifiable information.</p></fn><fn id="table1fn3"><p><sup>c</sup>ER: entity recognition.</p></fn><fn id="table1fn4"><p><sup>d</sup>NER: named entity recognition.</p></fn><fn id="table1fn5"><p><sup>e</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-8-2"><title>Data Sources</title><p>The primary inputs were group-session audio recordings (.wav or .mp3) recorded across multiple Shamiri Hub sites at three native sampling rates: 22,050 Hz (43.2%), 44,100 Hz (22.7%), and 48,000 Hz (34.1%). All audio was loaded for feature extraction at its native sampling rate without resampling; consequently, spectral features, mel-frequency cepstral coefficients (MFCCs), and pitch-tracking outputs were computed at the native rate per file and were not directly comparable across sessions recorded at different rates without normalization. ASR training and evaluation used a curated corpus of historical Shamiri recordings from 2023 to 2024, comprising approximately 10 hours across multiple sessions, with reference transcripts prepared by bilingual annotators familiar with Kenyan adolescent speech, capturing English, Kiswahili, and Sheng as spoken. The corpus was split into training (8 h), validation (1 h), and held-out test (1 h) sets. All ASR performance metrics reported in the Results section were based exclusively on the held-out test set. The 52 fidelity-validation sessions comprised 19 Session 1 (Growth Mindset I), 15 Session 2 (Growth Mindset II), 12 Session 3 (Gratitude), and 6 Session 4 (Values Affirmation) recordings.</p></sec><sec id="s2-8-3"><title>Stage 1: Ingestion and Preprocessing</title><p>Hub Coordinators uploaded recordings to secure cloud storage after each session, and a scheduled server-based pipeline retrieved new files. Preprocessing comprised (1) format normalization to 16-kHz mono for consistent downstream input, (2) denoising via spectral gating and noise reduction to mitigate background noise in open environments, and (3) voice activity detection (VAD) to segment audio into speech and nonspeech regions, reducing computation and capturing conversational structure.</p></sec><sec id="s2-8-4"><title>Stage 2: Core Processing</title><p>Stage 2 consisted of the following two major branches executed in parallel: (1) multilingual ASR with speaker diarization and (2) prosodic feature extraction.</p><sec id="s2-8-4-1"><title>Stage 2a. Multilingual ASR</title><sec id="s2-8-4-1-1"><title>Model Selection and Architecture</title><p>We used an open-source Whisper-based ASR model as the transcription backbone [<xref ref-type="bibr" rid="ref54">54</xref>], selected for three reasons: strong multilingual performance, including Kiswahili; open-source licensing; and on-premise deployability, which allowed us to avoid transmitting sensitive school recordings to third-party servers. We started from the Whisper Small checkpoint (~242 million parameters), which balances accuracy and computational efficiency for our deployment context.</p></sec><sec id="s2-8-4-1-2"><title>Fine-Tuning</title><p>The base model was fine-tuned on the 8-hour training split to improve performance on code-switched Kenyan adolescent speech using a learning rate of 1&#x00D7;10&#x207B;&#x2075;, batch size of 16 per device, and 5000 training steps, with data augmentation (time masking and additive noise) to improve robustness to recording-quality variation. A separate diarization module identified speaker change points and assigned local labels (eg, &#x201C;Speaker 1&#x201D; and &#x201C;Speaker 2&#x201D;), enabling per-speaker analytics such as talk-time distribution. VAD metadata were attached to each diarized segment to flag silence and identify potential hallucinations (output produced despite no detected voice activity).</p></sec><sec id="s2-8-4-1-3"><title>Computational Resources</title><p>Fine-tuning ran on a single NVIDIA Tesla T4 GPU (16 GB) using Google Colab in approximately 11 hours; the checkpoint was finalized on May 26, 2025. Inference over the held-out evaluation set completed in approximately 68 minutes on the same GPU. All other local processing, including audio preprocessing, prosodic feature extraction, and named entity recognition (NER)-based PII redaction, ran on a consumer laptop without dedicated GPU. LLM fidelity scoring used the hosted Google Gemini 2.5 Pro API (gemini-2.5-pro) and required no local GPU. All model development and inference were conducted between May 26, 2025 (fine-tuning completion) and July 15, 2025 (final inference run).</p></sec><sec id="s2-8-4-1-4"><title>Transcription Metrics</title><p>Because sessions involve code-switching and agglutinative morphology, particularly in Kiswahili, we evaluated ASR using word error rate (WER), character error rate (CER), and semantic similarity (cosine similarity between multilingual sentence embeddings) [<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>] (refer to Measures for definitions).</p></sec><sec id="s2-8-4-1-5"><title>Zero-Shot Baseline</title><p>To contextualize the fine-tuned Whisper Small model, we report zero-shot baseline figures from the published literature rather than running a within-corpus zero-shot evaluation. Zero-shot Whisper Medium performance on Kiswahili is not directly reported in the original release [<xref ref-type="bibr" rid="ref54">54</xref>] but scales predictably with per-language pretraining volume. Published Swahili fine-tuning work reports zero-shot WER of 0.51&#x2010;0.60 on read speech before domain adaptation [<xref ref-type="bibr" rid="ref57">57</xref>], and code-switched WER is expected to be substantially higher [<xref ref-type="bibr" rid="ref58">58</xref>].</p></sec><sec id="s2-8-4-1-6"><title>Speaker Diarization</title><p>Diarization used pyannote/speaker-diarization-3.1 (pyannote.ai) [<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref60">60</xref>], a fully automatic neural pipeline requiring no manual VAD, speaker count, or dataset-specific fine-tuning. It ingests mono 16-kHz audio, downmixing and resampling as needed. No domain adaptation or hyperparameter tuning was applied. Absent manually annotated reference diarizations for the 52 sessions, session-level diarization error rate (DER) was not computed on the Shamiri corpus. Published benchmark DER values for this pipeline across nine standard corpora (AISHELL-4, AliMeeting, AMI, AVA-AVD, DIHARD 3, MSDWild, REPERE, and VoxConverse) range from 7.8% to 50%, depending on recording conditions, with multispeaker meeting benchmarks (AMI and DIHARD 3) suggesting a plausible expected DER of 18%&#x2010;22%.</p></sec></sec></sec></sec><sec id="s2-9"><title>Stage 2b: Prosodic and Conversational Feature Extraction</title><p>In parallel with ASR, we extracted prosodic and conversational features using deterministic signal-processing techniques in librosa and related libraries. These features capture dimensions of group dynamics and facilitator behavior not reflected in transcribed text and were supplied as supplementary inputs to LLM inference [<xref ref-type="bibr" rid="ref61">61</xref>-<xref ref-type="bibr" rid="ref63">63</xref>].</p><sec id="s2-9-1"><title>Spectral Features</title><p>Spectral centroid (frequency &#x201C;center of mass&#x201D;; vocal brightness/energy), spectral bandwidth (spread of the frequency distribution; articulation clarity), and spectral contrast (expressiveness/vocal variation) [<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref64">64</xref>].</p></sec><sec id="s2-9-2"><title>Speech-Segment Features</title><p>Number of speech segments (turns), mean and SD of segment durations, mean and SD of interspeaker gaps, and quick-exchange ratio (proportion of gaps below a short-gap threshold), which served as proxies for interaction quality, equitable participation, pacing, and flow [<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref62">62</xref>].</p></sec><sec id="s2-9-3"><title>Voice Diversity</title><p>Pitch diversity (variation in fundamental frequency across speakers and time) and energy diversity (variation in vocal intensity); higher values may indicate distributed rather than facilitator-dominated participation [<xref ref-type="bibr" rid="ref65">65</xref>].</p></sec><sec id="s2-9-4"><title>Conversation-Flow Indicators</title><p>Speech density (speech vs. silence within sliding windows), MFCCs (timbral characteristics), and rhythm features (tempo and periodicity), which relate to session intensity and pacing [<xref ref-type="bibr" rid="ref66">66</xref>,<xref ref-type="bibr" rid="ref67">67</xref>].</p></sec><sec id="s2-9-5"><title>Temporal Dynamics</title><p>Trajectories of pitch and energy across the session and shifts in turn-taking over time, characterizing the therapeutic arc (opening, working phase, and closure) and moments of notable change in group energy [<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref64">64</xref>].</p></sec><sec id="s2-9-6"><title>Voice-Quality Indicators</title><p>Zero-crossing rate, harmonic-to-noise ratio, and approximations of jitter and shimmer, which may reflect vocal tension or emotional states signaling group discomfort or facilitator stress [<xref ref-type="bibr" rid="ref68">68</xref>,<xref ref-type="bibr" rid="ref69">69</xref>].</p><p>All features were computed at both segment and session level and normalized across sessions to support within- and between-lay provider comparisons.</p></sec></sec><sec id="s2-10"><title>Stage 3: Postprocessing and PII Scrubbing</title><p>Before transcripts reached the LLM, PII scrubbing used NuNerZero (numind/NuNerZero; NuMind), a zero-shot named entity recognition model based on the GLiNER architecture [<xref ref-type="bibr" rid="ref70">70</xref>]. NuNerZero is a compact bidirectional token classifier trained on the NuNER (version 2.0) dataset, selected for its zero-shot capability (ie, requiring no labeled target-domain data) and support for arbitrary entity-label prompting at inference without fine-tuning. Three entity types (person, location, and organization) were targeted, and detected entities were replaced in place with typed placeholders ([REDACTED_PERSON], [REDACTED_LOCATION], and [REDACTED_ORGANIZATION]). Redaction was applied to ASR transcripts before any content reached LLM inference; audio and unredacted transcripts were retained on secure servers with restricted access. An English BERT (Bidirectional Encoder Representations from Transformers)&#x2013;based NER model (dslim/bert-base-NER) was implemented as a fallback but not used in production. Formal precision and recall for the redaction pipeline were not computed in this pilot study.</p></sec><sec id="s2-11"><title>Stage 4: LLM-Based Feedback Inference</title><p>Structured feedback reports were generated using Google Gemini 2.5 Pro (gemini-2.5-pro), selected after benchmarking against several contemporary LLMs, including GPT-based models, because of its superior performance on code-switched transcripts. The model was invoked once per session, receiving the diarized transcript and session-level audio features as structured JSON. Generation hyperparameters were fixed identically across all 52 sessions (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement C8): temperature of 1.0, top-p of 0.95, top-k of 40, maximum output of 65,536 tokens, and a dynamic thinking budget. The model returned a strictly typed JSON object validated against a Pydantic schema, which enforced integer scores on the 1&#x2010;7 scale for each rubric dimension and prevented continuous or out-of-range values. No postprocessing or regex extraction was applied; scores were obtained directly from the validated object.</p><p>Because no random seed was fixed, generation was stochastic, though score-level variance across calls was bounded by the discrete 1&#x2010;7 schema. To confirm this empirically, we ran a test-retest analysis on all 52 sessions by running the production prompt 3 times per session and computing the intraclass correlation coefficient (ICC [3,1]) across runs for each dimension. Test-retest ICC values ranged from 0.44 (Q6: Protocol Boundaries) to 0.74 (Q5: Clarity and Accessibility), with an overall composite ICC of 0.71, comparable to the human-human interrater reliability observed in this dataset (refer to <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement C8.3 for full methods and results).</p><p>The structured system prompt (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement C9) provided (1) background on the Shamiri intervention, its session-by-session structure, and supervision goals; (2) the six fidelity domains (Required Contents, Specifics, Thoroughness, Clarity, Skill, and Purity), including the scoring rubric and behavioral anchors; (3) the redacted transcript and a session-level audio-features object; and (4) a request for structured feedback, including numerical ratings per domain, specific strengths, areas for improvement with transcript evidence, and suggested focus areas for the upcoming supervision session.</p><p>Prosodic features were passed as a raw JSON object with original numeric values (eg, spectral_centroid_mean: 2847.3; speech_gap_mean: 1.42); no bucketing, z-scoring, or label substitution was applied before prompt construction. The prompt provided an interpretation guide mapping each technical field to a plain-language description (eg, quick_exchange_ratio &#x2192; &#x201C;conversation flow and turn-taking patterns&#x201D;; pitch_diversity &#x2192; &#x201C;emotional expressiveness and vocal variation&#x201D;), instructing the model to use these descriptions rather than raw field names in its output. Per-segment prosodic features were embedded within each transcript turn object, keeping the acoustic signal colocated with the corresponding speech text in the model context. Whether prosodic features measurably improve fidelity scoring beyond transcript content alone was not isolated in this pilot and remains to be established.</p><p>The system prompt was developed iteratively across 3 versions using a development corpus of earlier Shamiri Hub recordings [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>], distinct from and nonoverlapping with the 52 validation sessions. Refinements were driven by qualitative review of model outputs on this corpus and by progressively closer alignment with the Shamiri Intervention Protocol rubric. The final prompt was locked before assembly and scoring of the validation dataset; no validation session was reviewed, scored, or consulted during prompt development. The 52 sessions therefore constituted a clean held-out evaluation of the locked prompt, with no leakage between development and validation. Prompt-development artifacts were committed to version control together at project handoff rather than incrementally; therefore, file-level timestamps do not individually reflect chronological order, and the provenance account rests on author attestation rather than automated artifact evidence.</p></sec><sec id="s2-12"><title>Stage 5: Delivery of Feedback to Supervisors</title><p>For each session, shamiriAI generated a PDF report containing session metadata (lay provider ID, school, session number, date, and duration), a quantitative summary of key prosodic indicators (eg, talk-time distribution, speech density, and quick-exchange ratio), numerical ratings per fidelity domain, and a narrative section structured around strengths, areas for improvement, and suggested focus areas. Reports were delivered by secure email or through a shared folder and reviewed during weekly supervision. Supervisors were encouraged to interpret AI-generated ratings and feedback alongside their own knowledge of lay providers and clinical judgment.</p></sec><sec id="s2-13"><title>Deployment and Integration</title><p>During the study, transcription and feedback generation were orchestrated via a Python pipeline, with human operators managing uploads and report distribution. Longer-term plans include integration into the Shamiri digital infrastructure (shamiriOS; Shamiri Institute) for automated ingestion, background processing, and in-platform feedback dashboards [<xref ref-type="bibr" rid="ref53">53</xref>].</p></sec><sec id="s2-14"><title>Measures</title><sec id="s2-14-1"><title>Transcription Performance</title><p>The primary source for ASR evaluation was a held-out test set of 10 sessions (~1 h) with manually prepared reference transcripts. Transcripts were prepared by bilingual annotators familiar with Kenyan adolescent speech, capturing code-switching among English, Kiswahili, and Sheng. Annotators followed a standardized protocol and resolved disagreements by consensus.</p><p>CER measures the Levenshtein edit distance at the character level, normalized by the total number of reference characters. CER was the primary error-rate metric because it is less sensitive to morphological variation than WER. For example, a single incorrect Kiswahili suffix counts as one character error but a full word error under WER, artificially inflating the latter [<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref71">71</xref>].</p><p>WER is the standard benchmark, calculations as (insertions + deletions + substitutions) &#x00F7; total reference words, and was reported as a secondary metric for comparability with published systems, with the caveat that it overpenalizes morphological variation in code-switched speech [<xref ref-type="bibr" rid="ref56">56</xref>].</p><p>Semantic similarity was assessed via cosine similarity between sentence-level embeddings from a multilingual model (LaBSE), capturing whether meaning was preserved despite orthographic variation [<xref ref-type="bibr" rid="ref72">72</xref>]. Recall-Oriented Understudy for Gisting Evaluation&#x2013;Longest Common Subsequence, as defined as the longest common subsequence between hypothesis and reference texts, was also computed [<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref74">74</xref>]. String similarity metrics, including Jaccard (token overlap) and Jaro-Winkler (character-level near-miss matching), were computed per segment and summarized as means and SDs. These metrics are well suited to the near-misspellings common in Sheng, where spelling is informal and variable [<xref ref-type="bibr" rid="ref56">56</xref>].</p><p>CER and semantic similarity were treated as the primary indicators and were used to monitor improvements during development. WER was reported for comparability. All metrics were computed using the held-out test set only.</p></sec><sec id="s2-14-2"><title>Fidelity Rating</title><p>Intervention fidelity was rated using a structured 6-domain instrument developed and used in prior Shamiri trials and routine quality monitoring [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. Each domain was rated on a 1&#x2010;7 scale (1=poor and 7=excellent):</p><list list-type="bullet"><list-item><p>Required Contents: the degree to which mandatory session elements (ie, those specified as nonnegotiable in the protocol, including the confidentiality explanation with the appropriate exception clause and required therapeutic techniques) were present and correctly executed.</p></list-item><list-item><p>Specifics: adherence to the content prescribed for that session number (ie, correct topic, activities, and booklet pages), independent of how thoroughly or skillfully it was delivered.</p></list-item><list-item><p>Thoroughness: depth and completeness of coverage, including whether the fellow allowed adequate time for reflection and discussion rather than rushing.</p></list-item><list-item><p>Clarity: whether content was communicated clearly, accessibly, and age-appropriately for secondary school students, including effective language mixing (English and Kiswahili or Sheng), analogies, and examples.</p></list-item><list-item><p>Skill: the fellow&#x2019;s use of protocol-specified facilitation techniques, including open-ended questioning, validation, rephrasing in students&#x2019; own words, connecting student experiences, and managing group dynamics.</p></list-item><list-item><p>Purity: the absence of content outside the Shamiri protocol, including off-topic discussion, unsolicited personal disclosures, and concepts not included in the manual.</p></list-item></list><p>For each of the 52 sessions, shamiriAI generated automated scores across all 6 domains via its LLM pipeline (Stage 4). Separately, 2 independent human raters rated each session. To prevent contamination, all human ratings were conducted by trained Shamiri supervisors drawn exclusively from hubs other than Ngong Hub; none had been assigned to the AI-augmented arm or had received or reviewed any shamiriAI feedback before or during rating. Raters were blind to AI scores and to each other&#x2019;s ratings. Each session&#x2019;s 2 human scores were averaged into a composite, which served as the human reference standard in all primary reliability analyses.</p></sec></sec><sec id="s2-15"><title>Statistical Analyses</title><sec id="s2-15-1"><title>Overview</title><p>Two sets of analyses aligned with the 2 pilot aims. Primary analyses were conducted in Python (version 3.12, Python Software Foundation; <italic>pandas</italic>, <italic>SciPy</italic>, and <italic>pingouin</italic>); Gwet AC2 with ordinal weights was computed via the <italic>irrCAC</italic> package, with categories inferred from observed values per the R <italic>irrCAC</italic> default to ensure cross-platform reproducibility. All analysis code is publicly available on the Open Science Framework. All 52 sessions had complete AI and human composite ratings across all 6 dimensions; no imputation or case exclusion was required.</p></sec><sec id="s2-15-2"><title>Transcription Performance</title><p>ASR was evaluated on the held-out test set using the multimetric approach described in the Measures section above (under Transcription Metrics). Segment-level metrics (Jaccard and Jaro-Winkler) are reported as means and SDs across segments. CER and semantic similarity are reported as primary indicators, with WER reported as a secondary comparability metric.</p></sec><sec id="s2-15-3"><title>Interrater Reliability</title><p>Primary analyses compared AI ratings against the human composite by computing the following for each dimension:</p><list list-type="order"><list-item><p>ICC: a 2-way random-effects, single-measures, absolute-agreement ICC estimated with pingouin [<xref ref-type="bibr" rid="ref75">75</xref>], treating both raters as random effects to permit generalization beyond the specific raters. ICCs are reported with 95% CIs and interpreted using established benchmarks: poor (&#x003C;0.50), moderate (0.50&#x2010;0.75), good (0.75&#x2010;0.90), excellent (&#x003E;0.90) [<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref77">77</xref>].</p></list-item><list-item><p>Agreement categories: per dimension, the percentage of sessions in exact agreement (difference=0), adjacent agreement (|difference|&#x2264;1), and discrepant (|difference|&#x003E;1).</p></list-item><list-item><p>Bland-Altman analysis: per dimension, plots of the difference (AI minus human composite) against the mean of the 2 methods, with mean difference (systematic bias) and 95% limits of agreement (SD 1.96), allowing visual inspection of directional bias and ranges of stronger or weaker agreement [<xref ref-type="bibr" rid="ref78">78</xref>].</p></list-item><list-item><p>Dimension-level bias: for each of the 6 dimensions and an Average Fidelity composite (mean of all 6 per session), we report means and SDs separately for AI and human composite. To test systematic differences, we conducted 2-tailed paired-sample <italic>t</italic> tests (<italic>&#x03B1;</italic>=.05) comparing AI against the human composite on each dimension and on Average Fidelity (7 tests), with Holm-Bonferroni correction across the 7 comparisons [<xref ref-type="bibr" rid="ref79">79</xref>]. We report mean differences (AI minus human composite), 95% CIs, <italic>t</italic> statistics, Holm-corrected <italic>P</italic> values, and Cohen <italic>d</italic>. Cohen <italic>d</italic> used the pooled SD of both methods rather than the SD of paired differences to express systematic bias relative to the natural variance of each measurement approach [<xref ref-type="bibr" rid="ref80">80</xref>]. Positive differences indicate that AI rated higher than the human composite; negative differences indicate lower ratings.</p></list-item></list></sec><sec id="s2-15-4"><title>Sensitivity Analyses</title><sec id="s2-15-4-1"><title>Gwet AC2</title><p>The shamiriAI model returns integer 1&#x2010;7 scores per dimension via schema-enforced output (Stage 4), while the human reference is the continuous mean of 2 raters. Our primary analysis uses the ICC (2-way random-effects, single-measures, absolute agreement) against the continuous human composite (Formulation B) consistent with conventional practice for agreement against continuous reference data. To address the asymmetry between the AI&#x2019;s discrete output and the continuous human reference, we added 2 sensitivity analyses using Gwet AC2 with ordinal weights (appropriate for ordinal ratings and robust to ceiling-concentrated distributions):</p><list list-type="bullet"><list-item><p>Formulation A: AI vs each individual human rater&#x2014;Gwet AC2 between AI integer scores and each rater&#x2019;s integer scores, preserving the discrete structure of both sources and avoiding compositing of the human reference.</p></list-item><list-item><p>Formulation C: AI vs rounded human composite&#x2014;Gwet AC2 between AI integer scores and the human composite rounded to the nearest integer, matching the AI&#x2019;s discrete scale to a discretized reference.</p></list-item></list><p>Dimension-level findings are interpreted as robust to the discrete-vs-continuous asymmetry if the ordering and sign of agreement are consistent between Formulation B (primary) and Formulations A and C (sensitivity). Full results across all 3 formulations appear in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement E2.</p></sec><sec id="s2-15-4-2"><title>Demographic Subgroup Analyses</title><p>To examine whether AI-human bias varied by lay-provider demographics, we stratified the per-session AI-minus-human-composite difference by sex (female/male) and fellow age band (median split at 20 years). Welch 2-sample <italic>t</italic> tests on the difference were conducted separately for each dimension, and Gwet AC2 (ordinal weights, against the rounded human composite) was computed within each stratum to characterize within-stratum agreement. These tests were exploratory, were not multiplicity corrected, and are reported descriptively rather than as confirmatory tests.</p></sec><sec id="s2-15-4-3"><title>Per-Arm Robustness Check</title><p>Because the 52 sessions were unevenly distributed between the AI-augmented (n=38) and standard (n=14) arms, we compared AI rating means, human composite means, and per-arm Gwet AC2 between arms for each dimension. Welch 2-sample <italic>t</italic> tests assessed between-arm differences in AI and human composite ratings; per-arm AC2 (ordinal weights, against the rounded human composite) characterized whether the dimension-level agreement pattern held within each arm. As with the subgroup analyses, these are exploratory and reported descriptively.</p></sec></sec></sec><sec id="s2-16"><title>Ethical Considerations</title><p>The study was approved by the Daystar University Institutional Scientific and Ethics Review Committee (DUISERC; approval DU-ISERC/10/06/2025/000015E; June 10, 2025) and licensed by the National Commission for Science, Technology and Innovation (NACOSTI; license NACOSTI/P/25/415077; issued February 11, 2025; valid through February 11, 2026). The trial was registered with the Pan African Clinical Trials Registry (PACTR202508900479778) on August 1, 2025. Audio recording at Ngong Hub began in May 2025, while routine program delivery operated under 2 existing approvals: the Kenyatta University Ethics Review Committee renewal (KUERC; application PKU/2627/E1752; renewed November 12, 2024; valid through November 12, 2025) for the broader &#x201C;Testing Pathways to Scale for a Cost-Effective and Evidence-Based Mental Health Innovation&#x201D; study, and the NACOSTI license (NACOSTI/P/25/415077) issued February 11, 2025 [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. The A/B-test&#x2013;specific DUISERC approval was received on June 10, 2025, and the first shamiriAI feedback reports were delivered to supervisors on June 20, 2025, after that approval was in place. PACTR registration was completed on August 1, 2025, after enrollment commenced; we acknowledge this as retrospective registration, which occurred because of in-house administrative transitions during a period of staff change. All data collection and AI-augmented supervision activities were conducted under existing DUISERC and KUERC approvals throughout. Written informed consent was obtained from all adult lay providers and adult student participants before involvement. For minors, written assent was obtained from the student, alongside parental or guardian consent secured through school administration, in accordance with DUISERC, KUERC, and NACOSTI protocols. Participants were informed that involvement was voluntary and that they could withdraw at any time without consequence. Lay providers were compensated at the standard Shamiri Institute rate of KES 1500 (approximately US $12) per 1-hour session delivered, consistent with their employment terms [<xref ref-type="bibr" rid="ref25">25</xref>]. Student participants received no compensation. Audio recordings and unredacted transcripts were stored on secure servers with restricted access. Before any AI inference, personally identifiable information (PII), including names, school identifiers, and location references, was detected and masked using a named entity recognition pipeline and replaced with typed placeholders ([REDACTED_PERSON], [REDACTED_LOCATION], and [REDACTED_ORGANIZATION]). Only deidentified, redacted transcripts were used in LLM-based fidelity inference (refer to Stage 3 and <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement C7). No participants are identified in any image or figure in this paper.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Sample Characteristics</title><p>Of the 64 Shamiri Fellows assigned to Ngong Hub, 47 participated. Thirty-four were randomized to the AI-augmented supervision arm and 13 to standard supervision (<xref ref-type="fig" rid="figure1">Figure 1</xref>; <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement D1). Most were female (36/47, 76.6%). Fellows ranged in age from 18 to 23 years (mean 19.83, SD 1.65). Across fellows, 52 recorded sessions were included in the fidelity analysis; most fellows contributed 1 or 2 recordings, and most rated sessions were from the AI-augmented arm (38/52, 73%). Seven supervisors participated; most were female (5/7, 71%), ranging in age from 23 to 30 years (mean 25.57, SD 2.23). Five (71%) supervisors were assigned to the AI-augmented arm and two (29%) to standard supervision. Sample characteristics are reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement D2.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>CONSORT (Consolidated Standards of Reporting Trials) diagram showing participant and session flow for the shamiriAI fidelity-validation pilot (conducted across 6 secondary schools at Ngong Hub, Kajiado County, Kenya, from May to September 2025). Lay providers were randomized to supervision arms; sessions were sampled independently of arm assignment for fidelity validation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95063_fig01.png"/></fig></sec><sec id="s3-2"><title>Aim 1: Transcription Performance</title><p>The fine-tuned Whisper model (small checkpoint;~242 million parameters) was evaluated on a held-out test set of 10 sessions with manually prepared reference transcripts. The model achieved a CER of 0.19 and a WER of 0.34. Cosine semantic similarity (LaBSE-based) was 0.77, and Recall-Oriented Understudy for Gisting Evaluation&#x2013;Longest Common Subsequence was 0.60. Segment-level Jaccard similarity averaged 0.72 (SD 0.26) and Jaro-Winkler similarity averaged 0.78 (SD 0.21). Full ASR metrics are provided in <xref ref-type="table" rid="table2">Table 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Automatic speech recognition performance metrics for the fine-tuned Whisper model on the held-out test set (n=10 sessions).</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom">Value</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Error-rate metrics</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Character error rate<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="top">0.19</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Word error rate<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">0.34</td></tr><tr><td align="left" valign="top" colspan="2">Semantic similarity metrics</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cosine semantic similarity (LaBSE)<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">0.77</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ROUGE-L<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">0.60</td></tr><tr><td align="left" valign="top" colspan="2">String similarity metrics (segment level)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Jaccard similarity</td><td align="left" valign="top">0.72</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Jaro-Winkler similarity</td><td align="left" valign="top">0.78</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Levenshtein edit distance at character level, normalized by total reference characters.</p></fn><fn id="table2fn2"><p><sup>b</sup>(insertions + deletions + substitutions) &#x00F7; total reference words.</p></fn><fn id="table2fn3"><p><sup>c</sup>LaBSE: Language-agnostic BERT (Bidirectional Encoder Representations from Transformers) Sentence Embedding; cosine similarity computed between sentence-level embeddings of hypothesis and reference transcripts.</p></fn><fn id="table2fn4"><p><sup>d</sup>ROUGE-L: Recall-Oriented Understudy for Gisting Evaluation&#x2013;Longest Common Subsequence variant.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Aim 2: Fidelity Rating</title><sec id="s3-3-1"><title>Overview</title><p>Fifty-two sessions were rated by both shamiriAI and 2 independent human raters drawn from trained Shamiri supervisors at hubs other than Ngong (refer to the Fidelity Rating subheading in the Methods section). Sessions spanned all four Shamiri session types across six schools: Session 1 (Growth Mindset I; n=19), Session 2 (Growth Mindset II; n=15), Session 3 (Gratitude; n=12), and Session 4 (Values Affirmation; n=6). All 52 sessions had complete AI and human scores across all 6 fidelity dimensions; no ratings were missing.</p><p><xref ref-type="table" rid="table3">Table 3</xref> provides descriptive statistics for AI and human composite ratings across all 6 dimensions and the Average Fidelity composite (Panel A in <xref ref-type="fig" rid="figure2">Figure 2</xref>). Human composite ratings ranged from mean 5.78 (SD 0.70) for Thoroughness to mean 6.14 (SD 0.74) for Required Contents, with SDs ranging from 0.58 to 0.74 across dimensions. AI ratings spanned a wider range, from mean 3.23 (SD 0.92) for Required Contents to mean 6.46 (SD 0.70) for Skill.</p><p>At the composite level, AI Average Fidelity (mean 5.14, SD 0.77) was lower than the human composite (mean 5.93, SD 0.57), with a mean difference of &#x2212;0.79 (95% CI &#x2212;1.04 to &#x2212;0.53; <italic>t</italic><sub>51</sub>=&#x2212;6.22; <italic>P</italic>&#x003C;.001; d=&#x2212;1.16) (Panel B in <xref ref-type="fig" rid="figure2">Figure 2</xref>).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Descriptive statistics, intraclass correlation coefficients, and agreement rates for AI vs human composite fidelity ratings (N=52).</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimension</td><td align="left" valign="bottom">AI ratings<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>, mean (SD)</td><td align="left" valign="bottom">AI ratings, range</td><td align="left" valign="bottom">Human composite<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>, mean (SD)</td><td align="left" valign="bottom">Human composite, range</td><td align="left" valign="bottom">ICC<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup> (95% CI)</td><td align="left" valign="bottom">Agreement (%), exact<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="bottom">Agreement (%), cumulative agreement<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup></td><td align="left" valign="bottom">Agreement (%), discrepant<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Required Contents</td><td align="left" valign="top">3.23 (0.92)</td><td align="left" valign="top">2&#x2010;6</td><td align="left" valign="top">6.14 (0.74)</td><td align="left" valign="top">5&#x2010;7</td><td align="left" valign="top">&#x2212;0.00 (&#x2212;0.03 to 0.05)</td><td align="left" valign="top">3.8</td><td align="left" valign="top">5.8</td><td align="left" valign="top">94.2</td></tr><tr><td align="left" valign="top">Specifics</td><td align="left" valign="top">5.54 (0.94)</td><td align="left" valign="top">3&#x2010;7</td><td align="left" valign="top">5.82 (0.71)</td><td align="left" valign="top">4&#x2010;7</td><td align="left" valign="top">0.03 (&#x2212;0.23 to 0.29)</td><td align="left" valign="top">13.5</td><td align="left" valign="top">78.8</td><td align="left" valign="top">21.2</td></tr><tr><td align="left" valign="top">Thoroughness</td><td align="left" valign="top">4.81 (1.21)</td><td align="left" valign="top">2&#x2010;7</td><td align="left" valign="top">5.78 (0.70)</td><td align="left" valign="top">4&#x2010;7</td><td align="left" valign="top">&#x2212;0.06 (&#x2212;0.23 to 0.15)</td><td align="left" valign="top">11.5</td><td align="left" valign="top">40.4</td><td align="left" valign="top">59.6</td></tr><tr><td align="left" valign="top">Clarity</td><td align="left" valign="top">4.50 (1.09)</td><td align="left" valign="top">2&#x2010;7</td><td align="left" valign="top">5.89 (0.64)</td><td align="left" valign="top">4&#x2010;7</td><td align="left" valign="top">0.14 (&#x2212;0.08 to 0.39)</td><td align="left" valign="top">3.8</td><td align="left" valign="top">40.4</td><td align="left" valign="top">59.6</td></tr><tr><td align="left" valign="top">Skill</td><td align="left" valign="top">6.46 (0.70)</td><td align="left" valign="top">4&#x2010;7</td><td align="left" valign="top">5.87 (0.67)</td><td align="left" valign="top">4&#x2010;7</td><td align="left" valign="top">0.12 (&#x2212;0.09 to 0.33)</td><td align="left" valign="top">25.0</td><td align="left" valign="top">71.2</td><td align="left" valign="top">28.8</td></tr><tr><td align="left" valign="top">Purity</td><td align="left" valign="top">6.31 (1.38)</td><td align="left" valign="top">2&#x2010;7</td><td align="left" valign="top">6.08 (0.58)</td><td align="left" valign="top">5&#x2010;7</td><td align="left" valign="top">0.20 (&#x2212;0.08 to 0.44)</td><td align="left" valign="top">15.4</td><td align="left" valign="top">73.1</td><td align="left" valign="top">26.9</td></tr><tr><td align="left" valign="top">Average Fidelity<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td><td align="left" valign="top">5.14 (0.77)</td><td align="left" valign="top">3.17&#x2010;6.67</td><td align="left" valign="top">5.93 (0.57)</td><td align="left" valign="top">4.58&#x2010;6.92</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn8">h</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>AI Ratings: shamiriAI large language model&#x2013;generated scores.</p></fn><fn id="table3fn2"><p><sup>b</sup>Human composite: mean of 2 independent human supervisor scores.</p></fn><fn id="table3fn3"><p><sup>c</sup>ICC: intraclass correlation coefficient (2-way random-effects, single-measure, absolute-agreement model).</p></fn><fn id="table3fn4"><p><sup>d</sup>Exact: sessions rated identically (difference=0).</p></fn><fn id="table3fn5"><p><sup>e</sup>Cumulative agreement: ratings within 1 scale point (|difference|&#x2264;1), inclusive of exact agreement.</p></fn><fn id="table3fn6"><p><sup>f</sup>Discrepant: ratings differing by &#x003E;1 scale point (|difference|&#x003E;1).</p></fn><fn id="table3fn7"><p><sup>g</sup>Average Fidelity: mean of all 6 dimension scores per session.</p></fn><fn id="table3fn8"><p><sup>h</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Distribution of shamiriAI vs human-composite mean fidelity ratings across the 6 rubric dimensions (Panel A) and the average fidelity composite (Panel B) (N=52 sessions). Ratings were on a 1&#x2010;7 scale (1=poor, 7=excellent); error bars represent &#x00B1;1 SEM. AI: shamiriAI large language model&#x2013;generated scores; human composite: mean of 2 independent human supervisor scores.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95063_fig02.png"/></fig></sec><sec id="s3-3-2"><title>Intraclass Correlation and Agreement</title><p>ICC estimates for AI vs human composite ratings (<xref ref-type="table" rid="table3">Table 3</xref>) ranged from &#x2212;0.06 (Thoroughness; 95% CI &#x2212;0.23 to 0.15) to 0.20 (Purity; 95% CI &#x2212;0.08 to 0.44). CIs for all 6 dimensions included 0, and all values fell below 0.50. Bland-Altman plots for all 6 dimensions are provided in <xref ref-type="fig" rid="figure3">Figure 3</xref>A. Limits of agreement (SD 1.96) ranged from approximately 2.4 scale points (Skill) to 5.0 scale points (Required Contents).</p><p>Adjacent agreement rates (|difference|&#x2264;1) varied by dimension: Specifics 78.8%, Purity 73.1%, Skill 71.2%, Thoroughness 40.4%, Clarity 40.4%, and Required Contents 5.8% (<xref ref-type="table" rid="table3">Table 3</xref>; Panel B in <xref ref-type="fig" rid="figure3">Figure 3</xref>). Exact agreement rates ranged from 3.8% (Required Contents and Clarity) to 25% (Skill). Discrepant ratings (|difference|&#x003E;1) ranged from 21.2% (Specifics) to 94.2% (Required Contents).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Bland-Altman plots and agreement-category distribution for shamiriAI vs human-composite fidelity ratings (N=52 sessions). (A) Bland-Altman plots for each of the 6 rubric dimensions, plotting the difference (AI minus human composite) against the mean of the 2 rating sources; the solid line indicates mean systematic bias, and dashed lines the 95% limits of agreement (SD 1.96). (B) stacked bar chart showing the percentage of sessions classified as Exact (difference=0), Adjacent (within 1 scale point but not exact), and Discrepant (&#x003E;1 scale point apart) for each dimension. AI: shamiriAI large language model&#x2013;generated scores; Human composite: mean of 2 independent human supervisor scores.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95063_fig03.png"/></fig></sec><sec id="s3-3-3"><title>Systematic Bias: Dimension-Level Analysis</title><p>Paired <italic>t</italic> tests comparing AI ratings against the human composite, with Holm-Bonferroni correction across 7 simultaneous comparisons, showed statistically significant systematic differences for 5 of 7 comparisons (<xref ref-type="table" rid="table4">Table 4</xref>; <xref ref-type="fig" rid="figure4">Figure 4</xref>).</p><p>AI underrated Required Contents (mean 3.23, SD 0.92 vs mean 6.14, SD 0.74; &#x0394;M=&#x2212;2.91, 95% CI &#x2212;3.25 to &#x2212;2.58; <italic>t</italic><sub>51</sub>=&#x2212;17.58; <italic>P</italic>&#x003C;.001; d=&#x2212;3.48), with 94.2% of sessions rated discrepantly (|difference|&#x003E;1). AI also underrated Clarity (&#x0394;M=&#x2212;1.39, 95% CI &#x2212;1.69 to &#x2212;1.10; <italic>t</italic><sub>51</sub>=&#x2212;9.58; <italic>P</italic>&#x003C;.001; d=&#x2212;1.56) and Thoroughness (&#x0394;M=&#x2212;0.97, 95% CI &#x2212;1.38 to &#x2212;0.57; <italic>t</italic><sub>51</sub>=&#x2212;4.83; <italic>P</italic>&#x003C;.001; d=&#x2212;0.99). AI overrated Skill (&#x0394;M=+0.60, 95% CI 0.35-0.84; <italic>t</italic><sub>51</sub>=4.85; <italic>P</italic>&#x003C;.001; d=+0.87). No statistically significant bias was observed for Specifics (&#x0394;M=&#x2212;0.28; <italic>P</italic>_adj=.18; d=&#x2212;0.34) or Purity (&#x0394;M=+0.23; <italic>P</italic>_adj=.22; d=+0.22); these dimensions had the highest adjacent agreement rates (78.8% and 73.1%, respectively).</p><p>For the Average Fidelity composite, AI ratings (mean 5.14, SD 0.77) were significantly lower than the human composite (mean 5.93, SD 0.57), with a mean difference of &#x2212;0.79 (95% CI &#x2212;1.04 to &#x2212;0.53; <italic>t</italic><sub>51</sub>=&#x2212;6.22; <italic>P</italic>&#x003C;.001; d=&#x2212;1.16) (<xref ref-type="fig" rid="figure4">Figure 4</xref>).</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Systematic bias analysis: paired <italic>t</italic> tests comparing AI and human composite fidelity ratings with Holm-Bonferroni correction (N=52). Effect size benchmarks: small |d| 0.20-0.49, medium 0.50-0.79, large &#x2265;0.80. ***<italic>P</italic>_adj&#x003C;.001. Mean differences (&#x0394;M) are computed from unrounded session-level scores and equal the mean of the paired (AI &#x2212; human composite) differences; they may differ from the difference of the rounded means shown by up to 0.01<italic>.</italic></p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimension</td><td align="left" valign="bottom">AI, mean (SD)</td><td align="left" valign="bottom">Human, mean (SD)</td><td align="left" valign="bottom">&#x0394;M<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> (95% CI)</td><td align="left" valign="bottom"><italic>t</italic> (<italic>df</italic>)</td><td align="left" valign="bottom"><italic>P</italic>_adj<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="bottom">d<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="bottom">Direction</td></tr></thead><tbody><tr><td align="left" valign="top">Required Contents</td><td align="left" valign="top">3.23 (0.92)</td><td align="left" valign="top">6.14 (0.74)</td><td align="left" valign="top">&#x2212;2.91 (&#x2212;3.25 to &#x2212;2.58)</td><td align="left" valign="top">&#x2212;17.58 (51)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x2212;3.48</td><td align="left" valign="top">Underrates<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">Specifics</td><td align="left" valign="top">5<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup>.54 (0.94)</td><td align="left" valign="top">5.82 (0.71)</td><td align="left" valign="top">&#x2212;0.28 (&#x2212;0.60 to 0.04)</td><td align="left" valign="top">&#x2212;1.74 (51)</td><td align="left" valign="top">.18</td><td align="left" valign="top">&#x2212;0.34</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td></tr><tr><td align="left" valign="top">Thoroughness</td><td align="left" valign="top">4.81 (1.21)</td><td align="left" valign="top">5.78 (0.70)</td><td align="left" valign="top">&#x2212;0.97 (&#x2212;1.38 to 0.57)</td><td align="left" valign="top">&#x2212;4.83 (51)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x2212;0.99</td><td align="left" valign="top">Underrates<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">Clarity</td><td align="left" valign="top">4.50 (1.09)</td><td align="left" valign="top">5.89 (0.64)</td><td align="left" valign="top">&#x2212;1.39 (&#x2212;1.69 to 1.10)</td><td align="left" valign="top">&#x2212;9.58 (51)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x2212;1.56</td><td align="left" valign="top">Underrates<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">Skill</td><td align="left" valign="top">6.46 (0.70)</td><td align="left" valign="top">5.87 (0.67)</td><td align="left" valign="top">0.60 (0.35-0.84)</td><td align="left" valign="top">4.85 (51)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.87</td><td align="left" valign="top">Overrates<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">Purity</td><td align="left" valign="top">6.31 (1.38)</td><td align="left" valign="top">6.08 (0.58)</td><td align="left" valign="top">0.23 (0.14-0.60)</td><td align="left" valign="top">1.24 (51)</td><td align="left" valign="top">.22</td><td align="left" valign="top">0.22</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Average Fidelity</td><td align="left" valign="top">5.14 (0.77)</td><td align="left" valign="top">5.93 (0.57)</td><td align="left" valign="top">&#x2212;0.79 (&#x2212;1.04 to 0.53)</td><td align="left" valign="top">&#x2212;6.22 (51)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x2212;1.16</td><td align="left" valign="top">Underrates<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>&#x0394;M: AI mean minus human composite mean (positive= AI rated higher; negative=AI rated lower).</p></fn><fn id="table4fn2"><p><sup>b</sup><italic>P</italic>_adj: Holm-Bonferroni-corrected <italic>P </italic>value across seven simultaneous comparisons (6 dimensions + Average Fidelity composite).</p></fn><fn id="table4fn3"><p><sup>c</sup><italic>d</italic>: Cohen <italic>d</italic> (pooled SD method).</p></fn><fn id="table4fn4"><p><sup>d</sup> shamiriAI either overrates or underrates human rates.</p></fn><fn id="table4fn5"><p><sup>e</sup>Nonsignificant.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Dimension-level systematic bias between shamiriAI and the human composite (N=52 sessions). The chart plots the mean difference (AI minus human composite) per rubric dimension with 95% CIs, alongside Cohen <italic>d</italic> effect sizes. Negative values indicate AI rated lower than the human composite; positive values indicate that AI rated higher. Effect sizes were computed using the pooled SD of the 2 rating methods. Asterisks indicate dimensions for which systematic bias was statistically significant after Holm-Bonferroni correction across 7 comparisons.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95063_fig04.png"/></fig></sec></sec><sec id="s3-4"><title>Secondary and Sensitivity Analyses</title><sec id="s3-4-1"><title>Human-Human Agreement</title><p>Human-human AC2 values were computed for all 5 rater pairs across 52 sessions. Human-human AC2 ranged from 0.42 (Purity) to 0.60 (Clarity), with all 6 dimensions in the moderate range (0.41&#x2010;0.60).</p></sec><sec id="s3-4-2"><title>AI-Human Agreement</title><p>Three formulations of AI-human agreement are reported (refer to Sensitivity Analyses in the Methods section). ICCs between AI ratings and the continuous human composite (Formulation B, primary) ranged from &#x2212;0.058 (Thoroughness) to 0.196 (Purity), with all 6 dimensions classified as &#x201C;poor&#x201D; according to Koo and Li benchmarks [87<xref ref-type="bibr" rid="ref81">81</xref>] (<xref ref-type="table" rid="table3">Table 3</xref>). Specifics and Purity reached 78.8% and 73.1% adjacent agreement, respectively (<xref ref-type="table" rid="table3">Table 3</xref>).</p><p>Gwet AC2 against each individual human rater (Formulation A, sensitivity) ranged from &#x2212;0.628 (Required Contents) to 0.764 (Specifics) for Rater 1 and from &#x2212;0.482 (Required Contents) to 0.645 (Skill) for Rater 2. Gwet AC2 against the rounded human composite (Formulation C, sensitivity) was &#x2212;0.667 (Required Contents), 0.687 (Specifics), 0.487 (Thoroughness), 0.376 (Clarity), 0.741 (Skill), and 0.762 (Purity). Full results across all 3 formulations are reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement E2.</p><p>AI-human AC2 differed across individual rater pairs for several dimensions. For Thoroughness, AC2 was 0.626 against Rater 1 and 0.360 against Rater 2; for Specifics, the corresponding values were 0.764 and 0.438.</p></sec><sec id="s3-4-3"><title>Demographic Subgroup Analyses</title><p>We computed AI-minus-human composite mean differences and Gwet AC2 stratified by Fellow sex (female: n=39 sessions and male: n=13 sessions) and fellow age band (median split at 20 years; &#x003C;20 years: n=23 sessions; &#x2265;20 years: n=29 sessions) for each of the 6 dimensions. Welch 2-sample <italic>t</italic> tests on AI-human differences yielded no statistically significant sex effect on any dimension (all <italic>P</italic>&#x003E;.25) and no statistically significant age-band effects on any dimension (all <italic>P</italic>&#x003E;.10). Per-stratum AC2 values are reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement E3 (Tables E3.1 and E3.2).</p></sec><sec id="s3-4-4"><title>Per-Arm Robustness Check</title><p>We compared AI rating means, human composite means, and per-arm Gwet AC2 between the AI-augmented (n=38 sessions) and standard supervision (n=14 sessions) arms for each of the 6 dimensions. AI rating means did not differ significantly between arms on any dimension (all Welch <italic>t</italic> test <italic>P</italic>&#x003E;.29), and human composite means also did not differ (all <italic>P</italic>&#x003E;.55). Per-arm AC2 followed the same dimension-dependent pattern as the overall analysis for 5 of 6 dimensions. Clarity AC2 was 0.394 in the AI-augmented arm and 0.034 in the standard arm, while AI rating means and human composite means for Clarity did not differ between arms (all <italic>P</italic>&#x003E;.61). Full per-arm distributions and AC2 estimates are reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>: Supplement E5.</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This pilot pursued 2 foundational aims for shamiriAI&#x2014;an automated fidelity-monitoring system for lay-delivered group mental health interventions in multilingual, low-resource settings.</p><p>The first aim established that the fine-tuned Whisper pipeline produced transcripts of sufficient semantic quality for downstream fidelity evaluation in naturalistic, multispeaker, code-switched English-Kiswahili-Sheng audio. A WER of 0.34 looks high against English-only benchmarks, but zero-shot Whisper on Kiswahili commonly exceeds 0.50 on read speech [<xref ref-type="bibr" rid="ref82">82</xref>,<xref ref-type="bibr" rid="ref83">83</xref>]. After being trained on about 8 hours of domain data and operating on spontaneous group speech, our model represents a meaningful gain at minimal cost. The metrics most appropriate for this context&#x2014;CER (0.19) and cosine semantic similarity (0.77)&#x2014;indicate that session meaning is preserved even where individual tokens are misrecognized. The output is semantically coherent enough to support LLM-based inference.</p><p>The second aim produced a more complex picture. Across all 6 dimensions, shamiriAI rated sessions an average of 0.79 points lower than the human panel on average&#x2014;a large aggregate effect (d=&#x2212;1.16) that remained significant after Holm-Bonferroni correction. The model scored more critically than human supervisors. That headline obscures 3 distinct patterns with different implications for development.</p></sec><sec id="s4-2"><title>Pattern 1: Limitations of Holistic Interpretive Dimensions</title><p>The largest discrepancies were observed for Required Contents (d=&#x2212;3.48) and Clarity (d= &#x2212;1.56). The AI differed from the panel by more than one scale point in 94% of Required Contents sessions, and Clarity adjacent agreement was 40%. Both dimensions require holistic, integrative judgment about whether a session was complete and clear &#x201C;in spirit,&#x201D; not whether scripted elements appeared verbatim. Human supervisors credit spirit-compliant delivery&#x2014;paraphrase, local idiom, and language mixing&#x2014;as meeting the standard, whereas the AI appears to apply a more literal rubric, penalizing missing scripted phrases [<xref ref-type="bibr" rid="ref84">84</xref>,<xref ref-type="bibr" rid="ref85">85</xref>]. Required Contents may have a further diagnosable cause: ASR errors in group audio destroy the short transitional utterances that signal element completion, and required elements are often spread across turns and speakers rather than delivered in one block. These are architectural, not prompt-tuning, problems and point to specific version 2 fixes, including required-element checklists in the prompt, sequential per-element evaluation before an overall rating, behavioral anchors from calibrated annotations, and few-shot examples spanning acceptable delivery styles.</p></sec><sec id="s4-3"><title>Pattern 2: Bidirectional Bias on Facilitation Dimensions</title><p>Two medium effects ran in opposite directions: the AI underrated Thoroughness (d=&#x2212;0.99) and overrated Skill (d=+0.87). The opposing signs suggest that the model reads surface markers of facilitation&#x2014;open questions, reflective language, validation, and restatement&#x2014;as skill [<xref ref-type="bibr" rid="ref85">85</xref>], but does not assess whether they were used at adequate depth, at the right moments, and with time for student reflection [<xref ref-type="bibr" rid="ref86">86</xref>]. A provider producing dense facilitation language while rushing content earns high Skill but low Thoroughness; a supervisor seeing both together rates them more consistently. Prompt redesign should partly correct this by incorporating Thoroughness anchors distinguishing depth from technique frequency and Skill anchors weighting contextual appropriateness alongside behavioral presence, without new training data.</p></sec><sec id="s4-4"><title>Pattern 3: Operationally Viable Performance on Structured Detection Dimensions</title><p>This is a positive finding. No significant bias was detected for Specifics (d=&#x2212;0.34; <italic>P</italic>=.18) or Purity (d=+0.22; <italic>P</italic>=.22), and adjacent agreement reached 78.8% and 73.1%, approaching the human&#x2013;human AC2 ceiling of 0.42&#x2010;0.60. Both are detection tasks&#x2014;whether prescribed content is present and prohibited content was absent&#x2014;and on these shamiriAI already performs at a level that could support supervisory decisions. This matters because Specifics and Purity are the dimensions most tied to protocol adherence and clinical safety, the quality floor that supervision exists to enforce. An AI system that flags off-protocol sessions and confirms coverage already gives supervisors actionable signals on what matters most for preventing drift [<xref ref-type="bibr" rid="ref86">86</xref>,<xref ref-type="bibr" rid="ref87">87</xref>].</p></sec><sec id="s4-5"><title>Interpreting the Reliability Indices</title><p>The reliability indices must be interpreted relative to 2 intrinsic properties of the human reference standard. First, human composite ratings clustered near the ceiling (mean 5.78&#x2010;6.14, SD 0.58&#x2010;0.74), restricting variance and mechanically suppressing ICCs regardless of true rank alignment. This explains why Specifics and Purity were classified as &#x201C;poor&#x201D; by ICC benchmarks [<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref88">88</xref>] yet exceeded 70% adjacent agreement. Gwet AC2, with prevalence- and bias-adjusted chance correction, partly corrects this and confirms the dimension ordering. Second, human-human AC2 ranged from 0.42 to 0.60, representing the realistic upper bound any model could achieve against this particular composite [<xref ref-type="bibr" rid="ref89">89</xref>]. AI-human AC2 on Specifics, Skill, and Purity approaches that ceiling rather than falling short of an unattainable maximum.</p><p>The 52-session sample is small for stable AC2 estimation, and CIs are wide (<xref ref-type="table" rid="table3">Tables 3 and 4</xref>). Accordingly, these estimates require replication in a planned larger cohort (target N=400). The direction is more secure: the dimension ordering&#x2014;highest for Specifics, Skill, and Purity; lowest for Required Contents; and intermediate for Thoroughness and Clarity&#x2014;held across the ICC analysis, both AC2 formulations, and the sex, age, and arm subgroups, suggesting that it is not an artifact of the reference treatment or demographic or arm-level confounding. These subgroup analyses were exploratory and underpowered, excluding only large effects.</p></sec><sec id="s4-6"><title>Comparison With Prior Work</title><p>Possibly comparable existing systems achieve higher reliability but under far more favorable conditions. Lyssn (Lyssn.io), a commercial platform trained on about 2500 labeled sessions of individual CBT by professional therapists in English, reaches 100% of human reliability on the overall quality score and exceeds 80% on 10 of 11 items [<xref ref-type="bibr" rid="ref36">36</xref>]. A recent LLM system scored patient engagement across 1131 individual CBT sessions in German with strong reliability and outcome-linked validity [<xref ref-type="bibr" rid="ref40">40</xref>]. shamiriAI currently achieves lower reliability, but the conditions differ fundamentally: individual vs group, professional vs lay providers, 1 language vs 3 including code-switched Sheng, large labeled corpora vs 8 hours, controlled vs open-environment audio. The relevant benchmark is the gap: on Specifics and Purity, shamiriAI already approaches the range of systems built for easier conditions, and its Required Contents and Clarity limitations are interpretable and addressable rather than fundamental. Procedural prompting has been shown to substantially improve LLM assessment reliability [<xref ref-type="bibr" rid="ref90">90</xref>], directly applicable to the priorities here.</p></sec><sec id="s4-7"><title>Limitations</title><p>This pilot has several limitations that constrain inference and set priorities for planned follow-up studies (target N=400) and future development.</p><p>First, the 52-sessions sample is small for stable agreement estimation in a ceiling-affected distribution; AC2 CIs are wide, and a larger sample is needed.</p><p>Second, the human reference showed only moderate interrater reliability (AC2 0.42&#x2010;0.60) and a restricted range (mean 5.78&#x2010;6.14, SD 0.58&#x2010;0.74), capping achievable agreement. Future iterations should incorporate structured rater calibration training and behavioral-anchored formats.</p><p>Third, the validation set was sampled without stratification by arm, producing a 38:14 imbalance. Per-arm comparisons detected no systematic bias, but validation with arm-balanced samples is needed.</p><p>Fourth, prompt-development provenance rests on author attestation rather than incremental timestamps, because artifacts were committed together at handoff; all 3 iterations are archived on Open Science Framework. Future development should adopt incremental version control.</p><p>Fifth, the PII redaction pipeline was not formally evaluated for precision and recall. Person-name redaction is qualitatively acceptable, but location and organization redaction is plausibly weaker for Sheng and Kiswahili. Accordingly, we treat NER as a supplementary layer behind secure storage and consent.</p><p>Sixth, audio recordings spanned 3 native sampling rates without resampling, so spectral features, MFCCs, and pitch outputs are not strictly comparable across sessions. Cross-rate normalization should be added in a v2.</p><p>Seventh, the zero-shot ASR baseline is drawn from published benchmarks [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref58">58</xref>] rather than a within-corpus run, so the fine-tuning gain cannot be precisely quantified; ASR metrics were also computed on a separate held-out test set, not the 52 fidelity validation sessions.</p><p>Eighth, diarization DER was not measured on the Shamiri corpus; we report published benchmark DER for pyannote [<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref60">60</xref>] as a proxy, and actual DER in adolescent group sessions may be worse.</p><p>Ninth, a prosodic-feature ablation was deferred to the larger cohort, where it can be adequately powered rather than interpreted post hoc.</p><p>Tenth, all sessions came from a single hub in Kajiado County; generalizability to other hubs, regions, languages of code-switching, or non-Shamiri interventions must be tested in multisite replication.</p><p>Eleventh, the trial was registered with PACTR after enrollment commenced; we acknowledge this departure from prospective registration and note that all activities fell within the DUISERC and KUERC approval period.</p><p>Twelfth, raters were calibrated before coding but did not recalibrate during the rating window and could not see one another&#x2019;s scores, so within-trial drift cannot be fully excluded.</p><p>Thirteenth, Session 4 was underrepresented (6/52, 11.5%), so its reliability should be reestimated in the larger cohort.</p><p>Fourteenth, scores come from a generative model sampled at temperature 1.0, so identical inputs can vary slightly across runs; test-retest agreement across 3 runs was high, and deterministic scoring would require temperature 0 or aggregation across runs.</p><p>Finally, this study evaluates technical performance only. Whether AI-augmented supervision improves provider skill, fidelity at scale, or student outcomes remains for the planned cluster-randomized noninferiority trial.</p></sec><sec id="s4-8"><title>Future Directions: shamiriAI in the Broader Vision</title><p>The goal of shamiriAI is not fidelity monitoring for its own sake but a quality-assurance infrastructure that can support lay-provider supervision at population scale&#x2014;the precondition for task-shifted care to reach the millions of young people who receive no help [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref91">91</xref>,<xref ref-type="bibr" rid="ref92">92</xref>]. This pilot establishes the 2 foundations required: a transcription pipeline that works in multilingual, code-switched, naturalistic group speech, and a fidelity-rating system whose failures are diagnosable rather than diffuse.</p><p>The immediate priority is shamiriAI (version 2)&#x2014;improved ASR through expanded fine-tuning, prompt redesign for Required Contents, Clarity, and Thoroughness, and integration into shamiriOS [<xref ref-type="bibr" rid="ref53">53</xref>] to deliver automated fidelity reports for every session at every active site, replacing the 10%&#x2010;15% of sessions that now receive any human review [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref93">93</xref>]. Once cross-dimension reliability reaches an adequate threshold, the next study is a fully powered cluster-randomized noninferiority trial comparing AI-augmented with standard supervision on youth depression and anxiety outcomes, provider skill, and cost-effectiveness. Beyond supervision, the session-level data shamiriAI generates can support causal mediation analyses, precision matching of students to providers, and identification of active ingredients [<xref ref-type="bibr" rid="ref94">94</xref>]&#x2014;a learning-system capacity that distinguishes it from a stand-alone monitoring tool and positions it as the backbone of an AI-native stepped-care model.</p><p>The work&#x2019;s importance rests on the supervision bottleneck. Kenya has about 2 specialized mental health workers per 100,000 people; sub-Saharan Africa has 1.4&#x2010;3.8 [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Task-shifting is the primary response to this gap and its effectiveness is well established [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref91">91</xref>], but supervision is the binding constraint on quality and scale [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref95">95</xref>,<xref ref-type="bibr" rid="ref96">96</xref>]. AI fidelity monitoring does not replace supervisors; it gives them the reach to supervise hundreds of providers across thousands of sessions. This pilot is the first evidence that such a system is technically buildable where it is most needed.</p></sec><sec id="s4-9"><title>Recommendations</title><p>The pilot specifies a concrete version 2 agenda: prompt redesign for Required Contents and Clarity with session-specific checklists and behavioral anchors, sequential per-element evaluation, Thoroughness and Skill anchors separating technique frequency from depth and timing, cross-rate feature normalization, and within-corpus evaluation of the zero-shot ASR baseline, diarization error, PII recall, and the prosodic ablation, all on the larger cohort. In parallel, future development should adopt incremental prompt version control with a dated changelog, prespecify arm-balanced sampling, audit PII redaction recall, and add calibration training for human raters. Beyond version 2, multihub replication and extension to non-Shamiri interventions and other low- and middle-income country contexts should inform version 3.</p></sec><sec id="s4-10"><title>Conclusions</title><p>Within a 52-session pilot, this study establishes 2 foundational results for shamiriAI. The fine-tuned multilingual Whisper pipeline achieved CER of 0.19 and cosine semantic similarity of 0.77 on naturalistic, multispeaker, code-switched English-Kiswahili-Sheng sessions, sufficient to support downstream LLM-based fidelity inference. The fidelity-rating system produced a dimension-dependent reliability profile: low ICCs against the continuous human composite (&#x2212;0.06 to 0.20), attributable to restriction of range, with AC2 sensitivity analyses corroborating substantial agreement on Specifics, Skill, and Purity (0.69&#x2010;0.76, approaching the human-human ceiling of 0.42&#x2010;0.60) and systematic underrating on Required Contents and Clarity. The results characterize current technical performance only; whether AI-augmented supervision improves provider skill, fidelity at scale, or student outcomes will be tested in the planned cluster-randomized noninferiority trial.</p></sec></sec></body><back><ack><p>The authors are grateful to the Hub staff, students, and schools that participated in this pilot. The authors used generative AI tools to assist with copyediting, formatting, and the preparation of figures during paper revision. All AI-assisted outputs were reviewed and verified by the authors, who take full responsibility for the content of the paper. Generative AI was not used to generate the study&#x2019;s scientific claims, results, or their interpretation.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the Johnson &#x0026; Johnson QuickFire Challenge and by accelerator funding from the Wellcome Trust. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the paper.</p></sec><sec><title>Data Availability</title><p>The fidelity-rating dataset, the deidentified transcription test set, and the analysis code (including the irrCAC scripts used for the agreement coefficients and the test-retest stability analysis) are available in the Open Science Framework repository for this project. Audio recordings and unredacted transcripts are not publicly available because they contain identifiable information about minors; deidentified derivatives may be requested from the corresponding author, subject to a data-use agreement and ethics approval.</p></sec></notes><fn-group><fn fn-type="con"><p>SL contributed to conceptualization, methodology, software, formal analysis, and writing of the original draft. BM contributed to software, data curation, investigation, and formal analysis. TO contributed to conceptualization, methodology, supervision, funding acquisition, and writing&#x2014;review and editing. WM contributed to investigation, data curation, and formal analysis. RK contributed to methodology, investigation, formal analysis, and writing&#x2014;review and editing. FK contributed to investigation, project administration, and writing&#x2014;review and editing. RD contributed to methodology, investigation, and writing&#x2014;review and editing. CW contributed to conceptualization, methodology, supervision, and writing&#x2014;review and editing.</p></fn><fn fn-type="conflict"><p>SL, BM, TO, WM, RK, FK, and RD are employees of Shamiri Institute, a nonprofit organization based in the Republic of Kenya. CW is a member of the Board of Directors of Shamiri Institute.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ASR</term><def><p>automatic speech recognition</p></def></def-item><def-item><term id="abb2">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb3">CBT</term><def><p>cognitive behavioral therapy</p></def></def-item><def-item><term id="abb4">CER</term><def><p>character error rate</p></def></def-item><def-item><term id="abb5">DER</term><def><p>diarization error rate</p></def></def-item><def-item><term id="abb6">DUISERC</term><def><p>Daystar University Institutional Scientific and Ethical Review Committee</p></def></def-item><def-item><term id="abb7">HIC</term><def><p>high-income country</p></def></def-item><def-item><term id="abb8">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb9">KUERC</term><def><p>Kenyatta University Ethics Review Committee</p></def></def-item><def-item><term id="abb10">LaBSE</term><def><p>Language-agnostic BERT Sentence Embedding</p></def></def-item><def-item><term id="abb11">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb12">MFCC</term><def><p>mel-frequency cepstral coefficient</p></def></def-item><def-item><term id="abb13">NACOSTI</term><def><p>National Commission for Science and Technology</p></def></def-item><def-item><term id="abb14">NER</term><def><p>named entity recognition</p></def></def-item><def-item><term id="abb15">PACTR</term><def><p>Pan African Clinical Trials Registry</p></def></def-item><def-item><term id="abb16">PII</term><def><p>personally identifiable information</p></def></def-item><def-item><term id="abb17">VAD</term><def><p>voice activity detection</p></def></def-item><def-item><term id="abb18">WER</term><def><p>word error rate</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="report"><person-group person-group-type="author"><collab>World Health Organization</collab></person-group><article-title>World mental health report: transforming mental health for all</article-title><year>2022</year><access-date>2022-06-17</access-date><publisher-name>World Health Organization</publisher-name><fpage>296</fpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/publications/i/item/9789240049338">https://www.who.int/publications/i/item/9789240049338</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Dou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Global, regional, and national burden of mental disorders among adolescents and young adults, 1990&#x2013;2021: a systematic analysis for the Global Burden of Disease Study 2021</article-title><source>Transl Psychiatry</source><year>2025</year><volume>15</volume><issue>1</issue><fpage>397</fpage><pub-id pub-id-type="doi">10.1038/s41398-025-03623-w</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>The Lancet</collab></person-group><article-title>Better understanding of youth mental health</article-title><source>The Lancet</source><year>2017</year><month>04</month><volume>389</volume><issue>10080</issue><fpage>1670</fpage><pub-id pub-id-type="doi">10.1016/S0140-6736(17)31140-6</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>GBD 2019 Mental Disorders Collaborators</collab></person-group><article-title>Global, regional, and national burden of 12 mental disorders in 204 countries and territories, 1990&#x2013;2019: a systematic analysis for the Global Burden of Disease Study 2019</article-title><source>Lancet Psychiatry</source><year>2022</year><month>02</month><volume>9</volume><issue>2</issue><fpage>137</fpage><lpage>150</lpage><pub-id pub-id-type="doi">10.1016/S2215-0366(21)00395-3</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kieling</surname><given-names>C</given-names> </name><name name-style="western"><surname>Buchweitz</surname><given-names>C</given-names> </name><name name-style="western"><surname>Caye</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Worldwide prevalence and disability from mental disorders across childhood and adolescence: evidence from the Global Burden of Disease study</article-title><source>JAMA Psychiatry</source><year>2024</year><month>04</month><day>1</day><volume>81</volume><issue>4</issue><fpage>347</fpage><lpage>356</lpage><pub-id pub-id-type="doi">10.1001/jamapsychiatry.2023.5051</pub-id><pub-id pub-id-type="medline">38294785</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Perera</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Child and adolescent mental health and psychosocial support interventions: an evidence and gap map of low&#x2010; and middle&#x2010;income countries</article-title><source>Campbell Syst Rev</source><year>2023</year><month>09</month><volume>19</volume><issue>3</issue><fpage>e1349</fpage><comment><ext-link ext-link-type="uri" xlink:href="https://onlinelibrary.wiley.com/toc/18911803/19/3">https://onlinelibrary.wiley.com/toc/18911803/19/3</ext-link></comment><pub-id pub-id-type="doi">10.1002/cl2.1349</pub-id><pub-id pub-id-type="medline">37621301</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jakobsson</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sanghavi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Nyamiobo</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Adolescent and youth-friendly health interventions in low-income and middle-income countries: a scoping review</article-title><source>BMJ Glob Health</source><year>2024</year><month>09</month><day>5</day><volume>9</volume><issue>9</issue><fpage>e013393</fpage><pub-id pub-id-type="doi">10.1136/bmjgh-2023-013393</pub-id><pub-id pub-id-type="medline">39242132</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="report"><article-title>Mental health atlas 2024</article-title><year>2024</year><access-date>2026-07-15</access-date><publisher-name>World Health Organization</publisher-name><fpage>98</fpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/publications/i/item/9789240114487">https://www.who.int/publications/i/item/9789240114487</ext-link></comment></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="report"><article-title>The old GHO Minerva interface and GHO Athena API are retired</article-title><year>2019</year><access-date>2020-04-24</access-date><publisher-name>WHO</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://apps.who.int/gho/data/node.main.MHHR?lang=en">https://apps.who.int/gho/data/node.main.MHHR?lang=en</ext-link></comment></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Baseke</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kamau</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Depression and anxiety symptoms among Kenyan adolescents: psychometric validation, prevalence, network analysis, and psychosocial determinants</article-title><source>Child Adolesc Psychiatry Ment Health</source><year>2026</year><month>06</month><day>16</day><pub-id pub-id-type="doi">10.1186/s13034-026-01117-1</pub-id><pub-id pub-id-type="medline">42304504</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osborn</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ndetei</surname><given-names>D</given-names> </name><name name-style="western"><surname>Mutiso</surname><given-names>V</given-names> </name></person-group><article-title>A &#x201C;both-and&#x201D; approach to cross-cultural mental health assessment: examining adolescent depression and anxiety symptoms using both local and western-derived instruments in Kenya</article-title><source>SSM - Mental Health</source><year>2026</year><month>12</month><volume>10</volume><fpage>100666</fpage><pub-id pub-id-type="doi">10.1016/j.ssmmh.2026.100666</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Venturo-Conerly</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Gan</surname><given-names>JY</given-names> </name><etal/></person-group><article-title>Depression and anxiety symptoms amongst Kenyan adolescents: psychometric properties, prevalence rates and associations with psychosocial wellbeing and sociodemographic factors</article-title><source>Res Child Adolesc Psychopathol</source><year>2022</year><month>11</month><volume>50</volume><issue>11</issue><fpage>1471</fpage><lpage>1485</lpage><pub-id pub-id-type="doi">10.1007/s10802-022-00940-2</pub-id><pub-id pub-id-type="medline">35675002</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Venturo-Conerly</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Wasil</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Schleider</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Weisz</surname><given-names>JR</given-names> </name></person-group><article-title>Depression and anxiety symptoms, social support, and demographic factors among Kenyan high school students</article-title><source>J Child Fam Stud</source><year>2020</year><month>05</month><volume>29</volume><issue>5</issue><fpage>1432</fpage><lpage>1443</lpage><pub-id pub-id-type="doi">10.1007/s10826-019-01646-8</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Meyer</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Ndetei</surname><given-names>D</given-names> </name></person-group><article-title>Providing sustainable mental health care in Kenya: a demonstration project</article-title><source>Providing Sustainable Mental and Neurological Health Care in Ghana and Kenya: Workshop Summary</source><year>2016</year><access-date>2026-03-08</access-date><publisher-name>National Academies Press (US)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/books/NBK350312">https://www.ncbi.nlm.nih.gov/books/NBK350312</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Wasanga</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Ndetei</surname><given-names>DM</given-names> </name></person-group><article-title>Transforming mental health for all</article-title><source>BMJ</source><year>2022</year><month>06</month><day>30</day><volume>377</volume><fpage>o1593</fpage><pub-id pub-id-type="doi">10.1136/bmj.o1593</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ndetei</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Mutiso</surname><given-names>VN</given-names> </name><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name></person-group><article-title>Moving away from the scarcity fallacy: three strategies to reduce the mental health treatment gap in LMICs</article-title><source>World Psychiatry</source><year>2023</year><month>02</month><volume>22</volume><issue>1</issue><fpage>163</fpage><lpage>164</lpage><pub-id pub-id-type="doi">10.1002/wps.21054</pub-id><pub-id pub-id-type="medline">36640407</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="web"><article-title>Task shifting: rational redistribution of tasks among health workforce teams: global recommendations and guidelines</article-title><source>World Health Organization</source><year>2008</year><access-date>2026-07-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://iris.who.int/handle/10665/43821">https://iris.who.int/handle/10665/43821</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bolton</surname><given-names>P</given-names> </name><name name-style="western"><surname>West</surname><given-names>J</given-names> </name><name name-style="western"><surname>Whitney</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Expanding mental health services in low- and middle-income countries: a task-shifting framework for delivery of comprehensive, collaborative, and community-based care</article-title><source>Glob Ment Health (Camb)</source><year>2023</year><volume>10</volume><fpage>e16</fpage><pub-id pub-id-type="doi">10.1017/gmh.2023.5</pub-id><pub-id pub-id-type="medline">37854402</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Joshi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Alim</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kengne</surname><given-names>AP</given-names> </name><etal/></person-group><article-title>Task shifting for non-communicable disease management in low and middle income countries &#x2013; a systematic review</article-title><source>PLoS ONE</source><year>2014</year><volume>9</volume><issue>8</issue><fpage>e103754</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0103754</pub-id><pub-id pub-id-type="medline">25121789</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bolton</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bass</surname><given-names>J</given-names> </name><name name-style="western"><surname>Neugebauer</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Group interpersonal psychotherapy for depression in rural Uganda: a randomized controlled trial</article-title><source>JAMA</source><year>2003</year><month>06</month><day>18</day><volume>289</volume><issue>23</issue><fpage>3117</fpage><lpage>3124</lpage><pub-id pub-id-type="doi">10.1001/jama.289.23.3117</pub-id><pub-id pub-id-type="medline">12813117</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Venturo-Conerly</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Arango G</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Effect of Shamiri layperson-provided intervention vs study skills control intervention for depression and anxiety symptoms in adolescents in Kenya: a randomized clinical trial</article-title><source>JAMA Psychiatry</source><year>2021</year><month>08</month><day>1</day><volume>78</volume><issue>8</issue><fpage>829</fpage><lpage>837</lpage><pub-id pub-id-type="doi">10.1001/jamapsychiatry.2021.1129</pub-id><pub-id pub-id-type="medline">34106239</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chibanda</surname><given-names>D</given-names> </name><name name-style="western"><surname>Weiss</surname><given-names>HA</given-names> </name><name name-style="western"><surname>Verhey</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Effect of a primary care-based psychological intervention on symptoms of common mental disorders in Zimbabwe: a randomized clinical trial</article-title><source>JAMA</source><year>2016</year><month>12</month><day>27</day><volume>316</volume><issue>24</issue><fpage>2618</fpage><lpage>2626</lpage><pub-id pub-id-type="doi">10.1001/jama.2016.19102</pub-id><pub-id pub-id-type="medline">28027368</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singla</surname><given-names>DR</given-names> </name><name name-style="western"><surname>Kohrt</surname><given-names>BA</given-names> </name><name name-style="western"><surname>Murray</surname><given-names>LK</given-names> </name><name name-style="western"><surname>Anand</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chorpita</surname><given-names>BF</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>V</given-names> </name></person-group><article-title>Psychological treatments for the world: lessons from low- and middle-income countries</article-title><source>Annu Rev Clin Psychol</source><year>2017</year><month>05</month><day>8</day><volume>13</volume><issue>1</issue><fpage>149</fpage><lpage>181</lpage><pub-id pub-id-type="doi">10.1146/annurev-clinpsy-032816-045217</pub-id><pub-id pub-id-type="medline">28482687</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Venturo-Conerly</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Wasil</surname><given-names>AR</given-names> </name><etal/></person-group><article-title>The Shamiri group intervention for adolescent anxiety and depression: study protocol for a randomized controlled trial of a lay-provider-delivered, school-based intervention in Kenya</article-title><source>Trials</source><year>2020</year><month>11</month><day>23</day><volume>21</volume><issue>1</issue><fpage>938</fpage><pub-id pub-id-type="doi">10.1186/s13063-020-04732-1</pub-id><pub-id pub-id-type="medline">33225978</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Venturo-Conerly</surname><given-names>K</given-names> </name><name name-style="western"><surname>Roe</surname><given-names>E</given-names> </name><name name-style="western"><surname>Wasil</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Training and supervising lay providers in Kenya: strategies and mixed-methods outcomes</article-title><source>Cogn Behav Pract</source><year>2022</year><month>08</month><volume>29</volume><issue>3</issue><fpage>666</fpage><lpage>681</lpage><pub-id pub-id-type="doi">10.1016/j.cbpra.2021.03.004</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kahi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Memba</surname><given-names>L</given-names> </name><name name-style="western"><surname>Syan</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Implementation of a school-based risk management protocol within a task-shifted mental healthcare model</article-title><source>Camb prisms Glob ment health</source><year>2025</year><volume>12</volume><fpage>e127</fpage><pub-id pub-id-type="doi">10.1017/gmh.2025.10073</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Venturo-Conerly</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Wasil</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Puffer</surname><given-names>ES</given-names> </name><name name-style="western"><surname>Weisz</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Wasanga</surname><given-names>CM</given-names> </name></person-group><article-title>Designing culturally and contextually sensitive protocols for suicide risk in global mental health: lessons from research with adolescents in Kenya</article-title><source>J Am Acad Child Adolesc Psychiatry</source><year>2022</year><month>09</month><volume>61</volume><issue>9</issue><fpage>1074</fpage><lpage>1077</lpage><pub-id pub-id-type="doi">10.1016/j.jaac.2022.02.005</pub-id><pub-id pub-id-type="medline">35217169</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Wasil</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Venturo-Conerly</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Schleider</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Weisz</surname><given-names>JR</given-names> </name></person-group><article-title>Group intervention for adolescent anxiety and depression: outcomes of a randomized trial with adolescents in Kenya</article-title><source>Behav Ther</source><year>2020</year><month>07</month><volume>51</volume><issue>4</issue><fpage>601</fpage><lpage>615</lpage><pub-id pub-id-type="doi">10.1016/j.beth.2019.09.005</pub-id><pub-id pub-id-type="medline">32586433</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Venturo-Conerly</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Rusch</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Testing the Shamiri intervention and its components with Kenyan adolescents during the COVID-19 pandemic: outcomes of a universal, 5-arm randomized controlled trial</article-title><source>J Am Acad Child Adolesc Psychiatry</source><year>2025</year><month>07</month><volume>64</volume><issue>7</issue><fpage>786</fpage><lpage>798</lpage><pub-id pub-id-type="doi">10.1016/j.jaac.2024.04.015</pub-id><pub-id pub-id-type="medline">38851382</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Venturo-Conerly</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Eisenman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wasil</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Singla</surname><given-names>DR</given-names> </name><name name-style="western"><surname>Weisz</surname><given-names>JR</given-names> </name></person-group><article-title>Meta-analysis: the effectiveness of youth psychotherapy interventions in low- and middle-income countries</article-title><source>J Am Acad Child Adolesc Psychiatry</source><year>2023</year><month>08</month><volume>62</volume><issue>8</issue><fpage>859</fpage><lpage>873</lpage><pub-id pub-id-type="doi">10.1016/j.jaac.2022.12.005</pub-id><pub-id pub-id-type="medline">36563875</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brooks</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Willis</surname><given-names>N</given-names> </name><name name-style="western"><surname>Beji-Chauke</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Principles for delivery of youth lay counsellor programs: Lessons from field experiences</article-title><source>J Glob Health</source><year>2022</year><month>07</month><day>25</day><volume>12</volume><issue>3047</issue><fpage>03047</fpage><pub-id pub-id-type="doi">10.7189/jogh.12.03047</pub-id><pub-id pub-id-type="medline">35871402</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Murray</surname><given-names>LK</given-names> </name><name name-style="western"><surname>Dorsey</surname><given-names>S</given-names> </name><name name-style="western"><surname>Haroz</surname><given-names>E</given-names> </name><etal/></person-group><article-title>A common elements treatment approach for adult mental health problems in low- and middle-income countries</article-title><source>Cogn Behav Pract</source><year>2014</year><month>05</month><volume>21</volume><issue>2</issue><fpage>111</fpage><lpage>123</lpage><pub-id pub-id-type="doi">10.1016/j.cbpra.2013.06.005</pub-id><pub-id pub-id-type="medline">25620867</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kohrt</surname><given-names>BA</given-names> </name><name name-style="western"><surname>Jordans</surname><given-names>MJD</given-names> </name><name name-style="western"><surname>Rai</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Therapist competence in global mental health: development of the ENhancing Assessment of Common Therapeutic factors (ENACT) rating scale</article-title><source>Behav Res Ther</source><year>2015</year><month>06</month><volume>69</volume><fpage>11</fpage><lpage>21</lpage><pub-id pub-id-type="doi">10.1016/j.brat.2015.03.009</pub-id><pub-id pub-id-type="medline">25847276</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wall</surname><given-names>JT</given-names> </name><name name-style="western"><surname>Kaiser</surname><given-names>BN</given-names> </name><name name-style="western"><surname>Friis-Healy</surname><given-names>EA</given-names> </name><name name-style="western"><surname>Ayuku</surname><given-names>D</given-names> </name><name name-style="western"><surname>Puffer</surname><given-names>ES</given-names> </name></person-group><article-title>What about lay counselors&#x2019; experiences of task-shifting mental health interventions? Example from a family-based intervention in Kenya</article-title><source>Int J Ment Health Syst</source><year>2020</year><volume>14</volume><fpage>9</fpage><pub-id pub-id-type="doi">10.1186/s13033-020-00343-0</pub-id><pub-id pub-id-type="medline">32099580</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meza</surname><given-names>RD</given-names> </name><name name-style="western"><surname>Kiche</surname><given-names>S</given-names> </name><name name-style="western"><surname>Soi</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Barriers and facilitators of child and guardian attendance in task-shifted mental health services in schools in western Kenya</article-title><source>Glob Ment Health</source><year>2020</year><volume>7</volume><pub-id pub-id-type="doi">10.1017/gmh.2020.9</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Creed</surname><given-names>TA</given-names> </name><name name-style="western"><surname>Salama</surname><given-names>L</given-names> </name><name name-style="western"><surname>Slevin</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Enhancing the quality of cognitive behavioral therapy in community mental health through artificial intelligence generated fidelity feedback (Project AFFECT): a study protocol</article-title><source>BMC Health Serv Res</source><year>2022</year><month>09</month><day>20</day><volume>22</volume><issue>1</issue><fpage>1177</fpage><pub-id pub-id-type="doi">10.1186/s12913-022-08519-9</pub-id><pub-id pub-id-type="medline">36127689</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ahmadi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Noetel</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schellekens</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A systematic review of machine learning for assessment and feedback of treatment fidelity</article-title><source>Psychosocial Intervention</source><year>2021</year><month>07</month><day>29</day><volume>30</volume><issue>3</issue><fpage>139</fpage><lpage>153</lpage><pub-id pub-id-type="doi">10.5093/pi2021a4</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Creed</surname><given-names>TA</given-names> </name><name name-style="western"><surname>Kuo</surname><given-names>PB</given-names> </name><name name-style="western"><surname>Oziel</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Knowledge and attitudes toward an artificial intelligence-based fidelity measurement in community cognitive behavioral therapy supervision</article-title><source>Adm Policy Ment Health</source><year>2022</year><month>05</month><volume>49</volume><issue>3</issue><fpage>343</fpage><lpage>356</lpage><pub-id pub-id-type="doi">10.1007/s10488-021-01167-x</pub-id><pub-id pub-id-type="medline">34537885</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moran</surname><given-names>LH</given-names> </name><name name-style="western"><surname>Kee</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Wiese</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Arriaga</surname><given-names>RI</given-names> </name><name name-style="western"><surname>Abdullah</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sherrill</surname><given-names>AM</given-names> </name></person-group><article-title>Artificial intelligence as a feedback teammate for treatment delivery: cognitive behavioral therapists&#x2019; hopes and fears</article-title><source>Cogn Behav Pract</source><year>2025</year><month>07</month><pub-id pub-id-type="doi">10.1016/j.cbpra.2025.06.007</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eberhardt</surname><given-names>ST</given-names> </name><name name-style="western"><surname>Vehlen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schaffrath</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Development and validation of large language model rating scales for automatically transcribed psychological therapy sessions</article-title><source>Sci Rep</source><year>2025</year><month>08</month><day>12</day><volume>15</volume><issue>1</issue><fpage>29541</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-14923-y</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ochuku</surname><given-names>B</given-names> </name><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Nerima</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Testing pathways to scale: study protocol for a three-arm randomized controlled trial of a centralized and a decentralized (&#x201C;Train the Trainers&#x201D;) dissemination of a mental health program for Kenyan adolescents</article-title><source>Trials</source><year>2023</year><month>08</month><day>13</day><volume>24</volume><issue>1</issue><fpage>526</fpage><pub-id pub-id-type="doi">10.1186/s13063-023-07539-y</pub-id><pub-id pub-id-type="medline">37574545</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baseke</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kilonzo</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ngesa</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mwende</surname><given-names>P</given-names> </name><name name-style="western"><surname>Osborn</surname><given-names>T</given-names> </name></person-group><article-title>A dataset on adolescent mental health in Kenya</article-title><source>Data Brief</source><year>2026</year><month>04</month><volume>65</volume><fpage>112513</fpage><pub-id pub-id-type="doi">10.1016/j.dib.2026.112513</pub-id><pub-id pub-id-type="medline">41732358</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Walton</surname><given-names>GM</given-names> </name></person-group><article-title>The new science of wise psychological interventions</article-title><source>Curr Dir Psychol Sci</source><year>2014</year><month>02</month><volume>23</volume><issue>1</issue><fpage>73</fpage><lpage>82</lpage><pub-id pub-id-type="doi">10.1177/0963721413512856</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Walton</surname><given-names>GM</given-names> </name><name name-style="western"><surname>Wilson</surname><given-names>TD</given-names> </name></person-group><article-title>Wise interventions: psychological remedies for social and personal problems</article-title><source>Psychol Rev</source><year>2018</year><month>10</month><volume>125</volume><issue>5</issue><fpage>617</fpage><lpage>655</lpage><pub-id pub-id-type="doi">10.1037/rev0000115</pub-id><pub-id pub-id-type="medline">30299141</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schleider</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Mullarkey</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Chacko</surname><given-names>A</given-names> </name></person-group><article-title>Harnessing wise interventions to advance the potency and reach of youth mental health services</article-title><source>Clin Child Fam Psychol Rev</source><year>2020</year><month>03</month><volume>23</volume><issue>1</issue><fpage>70</fpage><lpage>101</lpage><pub-id pub-id-type="doi">10.1007/s10567-019-00301-4</pub-id><pub-id pub-id-type="medline">31440858</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schleider</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Weisz</surname><given-names>JR</given-names> </name></person-group><article-title>Little treatments, promising effects? Meta-analysis of single-session interventions for youth psychiatric problems</article-title><source>J Am Acad Child Adolesc Psychiatry</source><year>2017</year><month>02</month><volume>56</volume><issue>2</issue><fpage>107</fpage><lpage>115</lpage><pub-id pub-id-type="doi">10.1016/j.jaac.2016.11.007</pub-id><pub-id pub-id-type="medline">28117056</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Mindset</surname><given-names>DCS</given-names> </name></person-group><source>The New Psychology of Success</source><year>2008</year><publisher-name>Random House Digital, Inc</publisher-name></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yeager</surname><given-names>DS</given-names> </name><name name-style="western"><surname>Dweck</surname><given-names>CS</given-names> </name></person-group><article-title>Mindsets that promote resilience: when students believe that personal characteristics can be developed</article-title><source>Educ Psychol</source><year>2012</year><month>10</month><volume>47</volume><issue>4</issue><fpage>302</fpage><lpage>314</lpage><pub-id pub-id-type="doi">10.1080/00461520.2012.722805</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Froh</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Sefick</surname><given-names>WJ</given-names> </name><name name-style="western"><surname>Emmons</surname><given-names>RA</given-names> </name></person-group><article-title>Counting blessings in early adolescents: an experimental study of gratitude and subjective well-being</article-title><source>J Sch Psychol</source><year>2008</year><month>04</month><volume>46</volume><issue>2</issue><fpage>213</fpage><lpage>233</lpage><pub-id pub-id-type="doi">10.1016/j.jsp.2007.03.005</pub-id><pub-id pub-id-type="medline">19083358</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Froh</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Kashdan</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Ozimkowski</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>N</given-names> </name></person-group><article-title>Who benefits the most from a gratitude intervention in children and adolescents? Examining positive affect as a moderator</article-title><source>J Posit Psychol</source><year>2009</year><month>09</month><volume>4</volume><issue>5</issue><fpage>408</fpage><lpage>422</lpage><pub-id pub-id-type="doi">10.1080/17439760902992464</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miyake</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kost-Smith</surname><given-names>LE</given-names> </name><name name-style="western"><surname>Finkelstein</surname><given-names>ND</given-names> </name><name name-style="western"><surname>Pollock</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>GL</given-names> </name><name name-style="western"><surname>Ito</surname><given-names>TA</given-names> </name></person-group><article-title>Reducing the gender achievement gap in college science: a classroom study of values affirmation</article-title><source>Science</source><year>2010</year><month>11</month><day>26</day><volume>330</volume><issue>6008</issue><fpage>1234</fpage><lpage>1237</lpage><pub-id pub-id-type="doi">10.1126/science.1195996</pub-id><pub-id pub-id-type="medline">21109670</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>GL</given-names> </name><name name-style="western"><surname>Garcia</surname><given-names>J</given-names> </name><name name-style="western"><surname>Purdie-Vaughns</surname><given-names>V</given-names> </name><name name-style="western"><surname>Apfel</surname><given-names>N</given-names> </name><name name-style="western"><surname>Brzustoski</surname><given-names>P</given-names> </name></person-group><article-title>Recursive processes in self-affirmation: intervening to close the minority achievement gap</article-title><source>Science</source><year>2009</year><month>04</month><day>17</day><volume>324</volume><issue>5925</issue><fpage>400</fpage><lpage>403</lpage><pub-id pub-id-type="doi">10.1126/science.1170769</pub-id><pub-id pub-id-type="medline">19372432</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lilan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Osborn</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Mmbone</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Early deployment of an integrated digital platform (shamiriOS) for scalable youth mental health service delivery in Kenya: development and usability study</article-title><source>JMIR Hum Factors</source><year>2026</year><month>06</month><day>3</day><volume>13</volume><fpage>e79107</fpage><pub-id pub-id-type="doi">10.2196/79107</pub-id><pub-id pub-id-type="medline">42235059</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Radford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Brockman</surname><given-names>G</given-names> </name><name name-style="western"><surname>Mcleavey</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sutskever</surname><given-names>I</given-names> </name></person-group><article-title>Robust speech recognition via large-scale weak supervision</article-title><access-date>2026-02-19</access-date><conf-name>Proceedings of the 40th International Conference on Machine Learning PMLR</conf-name><conf-date>Jul 23-29, 2023</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v202/radford23a.html">https://proceedings.mlr.press/v202/radford23a.html</ext-link></comment></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>K</surname><given-names>TD</given-names> </name><name name-style="western"><surname>James</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gopinath</surname><given-names>DP</given-names> </name><name name-style="western"><surname>K</surname><given-names>MA</given-names> </name></person-group><article-title>Advocating character error rate for multilingual ASR evaluation</article-title><conf-name>2025 Annual Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics</conf-name><conf-date>Apr 29 to May 4, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.findings-naacl.277</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Morris</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Maier</surname><given-names>V</given-names> </name><name name-style="western"><surname>Green</surname><given-names>PD</given-names> </name></person-group><article-title>From WER and RIL to MER and WIL: improved evaluation measures for connected speech recognition</article-title><year>2004</year><conf-name>Interspeech 2004</conf-name><fpage>2004</fpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.isca-archive.org/interspeech_2004/">https://www.isca-archive.org/interspeech_2004/</ext-link></comment><pub-id pub-id-type="doi">10.21437/Interspeech.2004-668</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Rouditchenko</surname><given-names>A</given-names> </name><name name-style="western"><surname>Khurana</surname><given-names>S</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Comparison of multilingual self-supervised and weakly-supervised speech pre-training for adaptation to unseen languages</article-title><source>arXiv</source><comment>Preprint posted online on  May 31, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2305.12606</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Peng</surname><given-names>P</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>B</given-names> </name><name name-style="western"><surname>Watanabe</surname><given-names>S</given-names> </name><name name-style="western"><surname>Harwath</surname><given-names>D</given-names> </name></person-group><article-title>Prompting the hidden talent of web-scale speech models for zero-shot task generalization</article-title><comment>Preprint posted online on  Aug 16, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2305.11095</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bredin</surname><given-names>H</given-names> </name></person-group><article-title>Pyannote.audio 2.1 speaker diarization pipeline: principle, benchmark, and recipe</article-title><year>2023</year><conf-name>INTERSPEECH 2023</conf-name><fpage>1983</fpage><lpage>1987</lpage><pub-id pub-id-type="doi">10.21437/Interspeech.2023-105</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Plaquet</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bredin</surname><given-names>H</given-names> </name></person-group><article-title>Powerset multi-class cross entropy loss for neural speaker diarization</article-title><year>2023</year><conf-name>INTERSPEECH 2023</conf-name><fpage>3222</fpage><lpage>3226</lpage><pub-id pub-id-type="doi">10.21437/Interspeech.2023-205</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Larrouy-Maestri</surname><given-names>P</given-names> </name><name name-style="western"><surname>Poeppel</surname><given-names>D</given-names> </name><name name-style="western"><surname>Pell</surname><given-names>MD</given-names> </name></person-group><article-title>The Sound of Emotional Prosody: Nearly 3 Decades of Research and Future Directions</article-title><source>Perspect Psychol Sci</source><year>2025</year><month>07</month><volume>20</volume><issue>4</issue><fpage>623</fpage><lpage>638</lpage><pub-id pub-id-type="doi">10.1177/17456916231217722</pub-id><pub-id pub-id-type="medline">38232303</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Klapprott</surname><given-names>F</given-names> </name><name name-style="western"><surname>K&#x00E4;stner</surname><given-names>D</given-names> </name><name name-style="western"><surname>Strau&#x00DF;</surname><given-names>B</given-names> </name><name name-style="western"><surname>Gumz</surname><given-names>A</given-names> </name></person-group><article-title>The role of prosody in therapists&#x2019; speech: A scoping review</article-title><source>Clin Psychol: Sci Pract</source><year>2024</year><volume>31</volume><issue>4</issue><fpage>508</fpage><lpage>523</lpage><pub-id pub-id-type="doi">10.1037/cps0000239</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Klapprott</surname><given-names>F</given-names> </name><name name-style="western"><surname>Strau&#x00DF;</surname><given-names>B</given-names> </name><name name-style="western"><surname>Gumz</surname><given-names>A</given-names> </name></person-group><article-title>More than words: The role of therapists&#x2019; prosody&#x2014;Reflections from practitioners</article-title><source>Psychol Psychother</source><year>2026</year><month>06</month><volume>99</volume><issue>2</issue><fpage>501</fpage><lpage>518</lpage><pub-id pub-id-type="doi">10.1111/papt.70033</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Soma</surname><given-names>CS</given-names> </name><name name-style="western"><surname>Knox</surname><given-names>D</given-names> </name><name name-style="western"><surname>Greer</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gunnerson</surname><given-names>K</given-names> </name><name name-style="western"><surname>Young</surname><given-names>A</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>S</given-names> </name></person-group><article-title>It&#x2019;s not what you said, it&#x2019;s how you said it: An analysis of therapist vocal features during psychotherapy</article-title><source>Couns Psychother Res</source><year>2023</year><month>03</month><volume>23</volume><issue>1</issue><fpage>258</fpage><lpage>269</lpage><pub-id pub-id-type="doi">10.1002/capr.12489</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yache</surname><given-names>VP</given-names> </name><name name-style="western"><surname>Moradbakhti</surname><given-names>L</given-names> </name><name name-style="western"><surname>Neuner</surname><given-names>I</given-names> </name><name name-style="western"><surname>Veselinovic</surname><given-names>T</given-names> </name></person-group><article-title>Predicting affective engagement and mental strain from prosodic speech features</article-title><source>Front Psychiatry</source><year>2025</year><volume>16</volume><fpage>1656292</fpage><pub-id pub-id-type="doi">10.3389/fpsyt.2025.1656292</pub-id><pub-id pub-id-type="medline">41048919</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rezapour Mashhadi</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Osei-Bonsu</surname><given-names>K</given-names> </name></person-group><article-title>Speech emotion recognition using machine learning techniques: feature extraction and comparison of convolutional neural network and random forest</article-title><source>PLoS ONE</source><year>2023</year><volume>18</volume><issue>11</issue><fpage>e0291500</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0291500</pub-id><pub-id pub-id-type="medline">37988352</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barhoumi</surname><given-names>C</given-names> </name><name name-style="western"><surname>BenAyed</surname><given-names>Y</given-names> </name></person-group><article-title>Real-time speech emotion recognition using deep learning and data augmentation</article-title><source>Artif Intell Rev</source><year>2024</year><month>12</month><day>20</day><volume>58</volume><issue>2</issue><fpage>49</fpage><pub-id pub-id-type="doi">10.1007/s10462-024-11065-x</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Teixeira</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Oliveira</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lopes</surname><given-names>C</given-names> </name></person-group><article-title>Vocal acoustic analysis &#x2013; jitter, shimmer and HNR parameters</article-title><source>Procedia Technology</source><year>2013</year><volume>9</volume><fpage>1112</fpage><lpage>1122</lpage><pub-id pub-id-type="doi">10.1016/j.protcy.2013.12.124</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kappen</surname><given-names>M</given-names> </name><name name-style="western"><surname>van der Donckt</surname><given-names>J</given-names> </name><name name-style="western"><surname>Vanhollebeke</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Acoustic speech features in social comparison: how stress impacts the way you sound</article-title><source>Sci Rep</source><year>2022</year><month>12</month><day>20</day><volume>12</volume><issue>1</issue><fpage>22022</fpage><pub-id pub-id-type="doi">10.1038/s41598-022-26375-9</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zaratiana</surname><given-names>U</given-names> </name><name name-style="western"><surname>Tomeh</surname><given-names>N</given-names> </name><name name-style="western"><surname>Holat</surname><given-names>P</given-names> </name><name name-style="western"><surname>Charnois</surname><given-names>T</given-names> </name></person-group><article-title>GLiNER: generalist model for named entity recognition using bidirectional transformer</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 14, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2311.08526</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Besacier</surname><given-names>L</given-names> </name><name name-style="western"><surname>Barnard</surname><given-names>E</given-names> </name><name name-style="western"><surname>Karpov</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schultz</surname><given-names>T</given-names> </name></person-group><article-title>Automatic speech recognition for under-resourced languages: A survey</article-title><source>Speech Commun</source><year>2014</year><month>01</month><volume>56</volume><fpage>85</fpage><lpage>100</lpage><pub-id pub-id-type="doi">10.1016/j.specom.2013.07.008</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>F</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cer</surname><given-names>D</given-names> </name><name name-style="western"><surname>Arivazhagan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>W</given-names> </name></person-group><article-title>Language-agnostic BERT sentence embedding</article-title><conf-name>60th Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>May 22-27, 2022</conf-date><pub-id pub-id-type="doi">10.18653/v1/2022.acl-long.62</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>CY</given-names> </name><name name-style="western"><surname>Och</surname><given-names>FJ</given-names> </name></person-group><article-title>Automatic evaluation of machine translation quality using longest common subsequence and skip-bigram statistics</article-title><conf-name>42nd Annual Meeting on Association for Computational Linguistics</conf-name><conf-date>Jul 21-26, 2004</conf-date><pub-id pub-id-type="doi">10.3115/1218955.1219032</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>CY</given-names> </name></person-group><article-title>ROUGE: a package for automatic evaluation of summaries</article-title><access-date>2026-07-16</access-date><conf-name>Proceedings of the Workshop on Text Summarization Branches Out (WAS 2004)</conf-name><conf-date>Jul 2004</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/volumes/W04-10/">https://aclanthology.org/volumes/W04-10/</ext-link></comment></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vallat</surname><given-names>R</given-names> </name></person-group><article-title>Pingouin: statistics in Python</article-title><source>JOSS</source><year>2018</year><volume>3</volume><issue>31</issue><fpage>1026</fpage><pub-id pub-id-type="doi">10.21105/joss.01026</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Gwet</surname><given-names>KL</given-names> </name></person-group><source>Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement among Raters</source><year>2014</year><publisher-name>Advanced Analytics, LLC</publisher-name></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gwet</surname><given-names>KL</given-names> </name></person-group><article-title>Computing inter-rater reliability and its variance in the presence of high agreement</article-title><source>Br J Math Stat Psychol</source><year>2008</year><month>05</month><volume>61</volume><issue>Pt 1</issue><fpage>29</fpage><lpage>48</lpage><pub-id pub-id-type="doi">10.1348/000711006X126600</pub-id><pub-id pub-id-type="medline">18482474</pub-id></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Martin Bland</surname><given-names>J</given-names> </name><name name-style="western"><surname>Altman</surname><given-names>D</given-names> </name></person-group><article-title>Statistical methods for assessing agreement between two methods of clinical measurement</article-title><source>The Lancet</source><year>1986</year><month>02</month><volume>327</volume><issue>8476</issue><fpage>307</fpage><lpage>310</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(86)90837-8</pub-id></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Holm</surname><given-names>S</given-names> </name></person-group><article-title>A simple sequentially rejective multiple test procedure</article-title><source>Scand J Stat</source><year>1979</year><access-date>2026-07-16</access-date><volume>6</volume><issue>2</issue><fpage>65</fpage><lpage>70</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.jstor.org/stable/4615733?origin=JSTOR-pdf">https://www.jstor.org/stable/4615733?origin=JSTOR-pdf</ext-link></comment></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Morris</surname><given-names>SB</given-names> </name></person-group><article-title>Estimating effect sizes from pretest-posttest-control group designs</article-title><source>Organ Res Methods</source><year>2008</year><month>04</month><volume>11</volume><issue>2</issue><fpage>364</fpage><lpage>386</lpage><pub-id pub-id-type="doi">10.1177/1094428106291059</pub-id></nlm-citation></ref><ref id="ref81"><label>81</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koo</surname><given-names>TK</given-names> </name><name name-style="western"><surname>Li</surname><given-names>MY</given-names> </name></person-group><article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research</article-title><source>J Chiropr Med</source><year>2016</year><month>06</month><volume>15</volume><issue>2</issue><fpage>155</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id><pub-id pub-id-type="medline">27330520</pub-id></nlm-citation></ref><ref id="ref82"><label>82</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sharma</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Pandya</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shukla</surname><given-names>A</given-names> </name></person-group><article-title>Fine-tuning whisper tiny for Swahili ASR: challenges and recommendations for low-resource speech recognition</article-title><conf-name>6th Workshop on African Natural Language Processing (AfricaNLP 2025)</conf-name><conf-date>Jul 31, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.africanlp-1.11</pub-id></nlm-citation></ref><ref id="ref83"><label>83</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Nahabwe</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kagumire</surname><given-names>S</given-names> </name><name name-style="western"><surname>Musinguzi</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Benchmarking automatic speech recognition models for African languages</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 30, 2025</comment><pub-id pub-id-type="doi">10.48550/ARXIV.2512.10968</pub-id></nlm-citation></ref><ref id="ref84"><label>84</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Evaluating an LLM-powered chatbot for cognitive restructuring: insights from mental health professionals</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 26, 2025</comment><pub-id pub-id-type="doi">10.48550/ARXIV.2501.15599</pub-id></nlm-citation></ref><ref id="ref85"><label>85</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Du</surname><given-names>L</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>R</given-names> </name></person-group><article-title>Evaluating large language models as raters in large-scale writing assessments: a psychometric framework for reliability and validity</article-title><source>Computers and Education: Artificial Intelligence</source><year>2025</year><month>12</month><volume>9</volume><fpage>100481</fpage><pub-id pub-id-type="doi">10.1016/j.caeai.2025.100481</pub-id></nlm-citation></ref><ref id="ref86"><label>86</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tan</surname><given-names>KS</given-names> </name><name name-style="western"><surname>Cervin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Leman</surname><given-names>P</given-names> </name><name name-style="western"><surname>Nielsen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>PV</given-names> </name><name name-style="western"><surname>Medvedev</surname><given-names>O</given-names> </name></person-group><article-title>AI meets psychology: an exploratory study of large language models&#x2019; competence in psychotherapy contexts</article-title><source>J Psychol AI</source><year>2025</year><month>12</month><day>31</day><volume>1</volume><issue>1</issue><fpage>2545258</fpage><pub-id pub-id-type="doi">10.1080/29974100.2025.2545258</pub-id></nlm-citation></ref><ref id="ref87"><label>87</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dun</surname><given-names>C</given-names> </name><name name-style="western"><surname>Couperus</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Evaluation of large language model performance in assessing health economic study quality</article-title><source>J Health Econ Outcomes Res</source><year>2025</year><volume>12</volume><issue>2</issue><fpage>154</fpage><lpage>162</lpage><pub-id pub-id-type="doi">10.36469/001c.145214</pub-id><pub-id pub-id-type="medline">41146963</pub-id></nlm-citation></ref><ref id="ref88"><label>88</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McGraw</surname><given-names>KO</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>SP</given-names> </name></person-group><article-title>Forming inferences about some intraclass correlation coefficients</article-title><source>Psychol Methods</source><year>1996</year><volume>1</volume><issue>1</issue><fpage>30</fpage><lpage>46</lpage><pub-id pub-id-type="doi">10.1037/1082-989X.1.1.30</pub-id></nlm-citation></ref><ref id="ref89"><label>89</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Landis</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Koch</surname><given-names>GG</given-names> </name></person-group><article-title>The measurement of observer agreement for categorical data</article-title><source>Biometrics</source><year>1977</year><month>03</month><volume>33</volume><issue>1</issue><fpage>159</fpage><pub-id pub-id-type="doi">10.2307/2529310</pub-id></nlm-citation></ref><ref id="ref90"><label>90</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lubbe</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wedderhoff</surname><given-names>N</given-names> </name><name name-style="western"><surname>Nelles</surname><given-names>C</given-names> </name></person-group><article-title>AI-driven versus human evaluations of psychotherapeutic communication performance: The impact of procedural prompting on assessment reliability and validity</article-title><source>Comput Hum Behav Rep</source><year>2026</year><month>03</month><volume>21</volume><fpage>100910</fpage><pub-id pub-id-type="doi">10.1016/j.chbr.2025.100910</pub-id></nlm-citation></ref><ref id="ref91"><label>91</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hoeft</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Fortney</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>V</given-names> </name><name name-style="western"><surname>Un&#x00FC;tzer</surname><given-names>J</given-names> </name></person-group><article-title>Task-sharing approaches to improve mental health care in rural and other low-resource settings: a systematic review</article-title><source>J Rural Health</source><year>2018</year><month>12</month><volume>34</volume><issue>1</issue><fpage>48</fpage><lpage>62</lpage><pub-id pub-id-type="doi">10.1111/jrh.12229</pub-id><pub-id pub-id-type="medline">28084667</pub-id></nlm-citation></ref><ref id="ref92"><label>92</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kemp</surname><given-names>CG</given-names> </name><name name-style="western"><surname>Petersen</surname><given-names>I</given-names> </name><name name-style="western"><surname>Bhana</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rao</surname><given-names>D</given-names> </name></person-group><article-title>Supervision of Task-shared mental health care in low-resource settings: a commentary on programmatic experience</article-title><source>Glob Health Sci Pract</source><year>2019</year><month>06</month><volume>7</volume><issue>2</issue><fpage>150</fpage><lpage>159</lpage><pub-id pub-id-type="doi">10.9745/GHSP-D-18-00337</pub-id><pub-id pub-id-type="medline">31249017</pub-id></nlm-citation></ref><ref id="ref93"><label>93</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bond</surname><given-names>L</given-names> </name><name name-style="western"><surname>Simmons</surname><given-names>E</given-names> </name><name name-style="western"><surname>Sabbath</surname><given-names>EL</given-names> </name></person-group><article-title>Measurement and assessment of fidelity and competence in nonspecialist-delivered, evidence-based behavioral and mental health interventions: A systematic review</article-title><source>SSM Popul Health</source><year>2022</year><month>09</month><volume>19</volume><fpage>101249</fpage><pub-id pub-id-type="doi">10.1016/j.ssmph.2022.101249</pub-id></nlm-citation></ref><ref id="ref94"><label>94</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Billovits</surname><given-names>C</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>SY</given-names> </name><etal/></person-group><article-title>Using machine learning to match clients and therapy providers: evaluating clinical quality and cost of care</article-title><source>Value Health</source><year>2025</year><month>09</month><volume>28</volume><issue>9</issue><fpage>1327</fpage><lpage>1334</lpage><pub-id pub-id-type="doi">10.1016/j.jval.2025.06.002</pub-id><pub-id pub-id-type="medline">40581153</pub-id></nlm-citation></ref><ref id="ref95"><label>95</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dorsey</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gray</surname><given-names>CL</given-names> </name><name name-style="western"><surname>Wasonga</surname><given-names>AI</given-names> </name><etal/></person-group><article-title>Advancing successful implementation of task-shifted mental health care in low-resource settings (BASIC): protocol for a stepped wedge cluster randomized trial</article-title><source>BMC Psychiatry</source><year>2020</year><month>01</month><day>8</day><volume>20</volume><issue>1</issue><fpage>10</fpage><pub-id pub-id-type="doi">10.1186/s12888-019-2364-4</pub-id><pub-id pub-id-type="medline">31914959</pub-id></nlm-citation></ref><ref id="ref96"><label>96</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barnett</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Gonzalez</surname><given-names>A</given-names> </name><name name-style="western"><surname>Miranda</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chavira</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Lau</surname><given-names>AS</given-names> </name></person-group><article-title>Mobilizing community health workers to address mental health disparities for underserved populations: a systematic review</article-title><source>Adm Policy Ment Health</source><year>2018</year><month>03</month><volume>45</volume><issue>2</issue><fpage>195</fpage><lpage>211</lpage><pub-id pub-id-type="doi">10.1007/s10488-017-0815-0</pub-id><pub-id pub-id-type="medline">28730278</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary materials.</p><media xlink:href="ai_v5i1e95063_app1.pdf" xlink:title="PDF File, 657 KB"/></supplementary-material><supplementary-material id="app2"><label>Checklist 1</label><p>TRIPOD+LLM checklist.</p><media xlink:href="ai_v5i1e95063_app2.pdf" xlink:title="PDF File, 68 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 2</label><p>CONSORT-eHEALTH checklist (V 1.6.1).</p><media xlink:href="ai_v5i1e95063_app3.pdf" xlink:title="PDF File, 1304 KB"/></supplementary-material></app-group></back></article>