<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e95964</article-id><article-id pub-id-type="doi">10.2196/95964</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Automating Motivational Interviewing Coding in Adolescent Substance Use Prevention: Human-AI Agreement Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Cardozo</surname><given-names>Francisco</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Brown</surname><given-names>Eric C</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mej&#x00ED;a-Trujillo</surname><given-names>Juliana</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Balise</surname><given-names>Raymond</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ca&#x00F1;izares</surname><given-names>Catalina</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>P&#x00E9;rez-G&#x00F3;mez</surname><given-names>Augusto</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>St George</surname><given-names>Sara M</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gabbay</surname><given-names>Vilma</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>Miller School of Medicine, University of Miami</institution><addr-line>1252 Memorial Dr, Coral Gables</addr-line><addr-line>Miami</addr-line><addr-line>FL</addr-line><country>United States</country></aff><aff id="aff2"><institution>Corporaci&#x00F3;n Nuevos Rumbos</institution><addr-line>Bogot&#x00E1;</addr-line><country>Colombia</country></aff><aff id="aff3"><institution>Applied Psychology, New York University</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff4"><institution>Nathan Kline Institute for Psychiatric Research</institution><addr-line>Orangeburg</addr-line><addr-line>NY</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Carcone</surname><given-names>April</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Wang</surname><given-names>Zhongyan</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Francisco Cardozo, PhD, Miller School of Medicine, University of Miami, 1252 Memorial Dr, Coral Gables, Miami, FL, 33146, United States, 1 305-284-2211; <email>foc9@miami.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>25</day><month>8</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e95964</elocation-id><history><date date-type="received"><day>26</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>22</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>22</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Francisco Cardozo, Eric C Brown, Juliana Mej&#x00ED;a-Trujillo, Raymond Balise, Catalina Ca&#x00F1;izares, Augusto P&#x00E9;rez-G&#x00F3;mez, Sara M St George, Vilma Gabbay. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 25.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e95964"/><abstract><sec><title>Background</title><p>Motivational interviewing (MI) is widely used in preventive interventions, yet coding MI techniques and monitoring intervention adherence remain resource-intensive due to the reliance on manual transcription and expert review. Large language models (LLMs) offer a promising approach to automate these tasks, but their agreement with human coders in the context of prevention interventions has not been established.</p></sec><sec><title>Objective</title><p>This study evaluated the agreement between an AI-based coder (OpenAI&#x2019;s GPT 4.1) and trained human coders on two tasks: (1) identification of MI techniques (eg, open questions, affirmations, giving information) at the facilitator-message level and (2) completing a 21-item checklist of implementation adherence for a brief MI-based preventive intervention for adolescent substance use.</p></sec><sec sec-type="methods"><title>Methods</title><p>Two certified MI facilitators independently coded 72 facilitator messages from 2 standardized Spanish-language Brief Intervention Based on Motivational Interviewing program (Intervenci&#x00F3;n Breve Basada en Entrevista Motivacional [IBEM]) practice sessions with chatbot-simulated adolescent responses. The facilitators classified MI techniques using the OARS (open questions, affirmations, reflections, and summaries) framework and completed a 21-item implementation-adherence checklist. An AI-based coder (OpenAI&#x2019;s GPT-4.1, accessed through the API) classified the same facilitator messages and checklist items using a structured prompt derived from the MI coding manual. MI techniques were compared at the facilitator-message level and implementation adherence at the session level. Intercoder agreement in use of MI techniques and implementation adherence was assessed using Cohen &#x03BA;, Fleiss &#x03BA;, and Cochran <italic>Q</italic> tests.</p></sec><sec sec-type="results"><title>Results</title><p>For use of MI techniques, the AI coder demonstrated moderate-to-substantial agreement with human coders across most techniques, including open questions (&#x03BA;=0.66-0.69), affirmations (&#x03BA;=0.66-0.77), and giving information (&#x03BA;=0.91). No statistically significant differences in percentages of MI technique use were observed among the 3 coders, although agreement was the weakest for higher-inference categories such as complex reflections (&#x03BA;=0.00). For MI implementation adherence, overall agreement was moderate (Fleiss &#x03BA;=0.487), and pairwise agreement between the AI coder and 1 human coder was substantial (&#x03BA;=0.67), exceeding the agreement observed between the 2 human coders (&#x03BA;=0.53).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>These findings provide support for the feasibility of using LLMs to recognize MI techniques and assess implementation adherence. The results support a human-AI collaborative model in which the AI coder &#x201C;precodes&#x201D; facilitator messages and flags sessions for expert review, while human coders retain responsibility for higher-inference judgments and shift their effort from routine coding toward contextual review and coaching feedback. Because the analyses are based on only 2 sessions, these results should be interpreted as early-stage, proof-of-concept evidence rather than a basis for large-scale deployment. Future research should compare different LLMs and evaluate whether AI-assisted coding improves the scalability of routine implementation monitoring.</p></sec></abstract><kwd-group><kwd>motivational interviewing</kwd><kwd>adherence</kwd><kwd>implementation fidelity</kwd><kwd>large language models</kwd><kwd>artificial intelligence</kwd><kwd>preventive interventions</kwd><kwd>intercoder agreement</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Motivational interviewing (MI) is a client-centered therapeutic approach designed to strengthen an individual&#x2019;s motivation and commitment to behavior change [<xref ref-type="bibr" rid="ref1">1</xref>]. In prevention programs, MI is often used to encourage positive health behaviors and reduce risk factors before problems escalate. A growing body of evidence supports MI&#x2019;s ability to enhance program engagement [<xref ref-type="bibr" rid="ref2">2</xref>], promote adherence [<xref ref-type="bibr" rid="ref3">3</xref>], and sustain healthier behaviors across diverse populations [<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>To implement MI effectively, facilitators train communication competencies such as reflective listening, empathetic communication, and the ability to guide individuals through ambivalence [<xref ref-type="bibr" rid="ref5">5</xref>]. These competencies are expressed in conversation through the use of open-ended questions, affirmations, reflective listening, and summaries, commonly referred to by the acronym OARS [<xref ref-type="bibr" rid="ref1">1</xref>]. OARS are fundamental to MI because they help participants articulate their experiences, recognize their strengths, feel heard, and engage in purposeful reflection, ultimately fostering motivation and promoting healthier behavioral choices. However, developing proficiency in these skills requires extensive practice through realistic conversational scenarios, and maintaining fidelity over time remains a persistent challenge in real-world settings [<xref ref-type="bibr" rid="ref6">6</xref>]. As a result, there is a growing need for scalable training and feedback systems that can support MI facilitators in refining and sustaining their delivery skills beyond initial certification.</p><p>Several methods have been developed to evaluate how facilitators can effectively deliver MI competencies. Two widely used approaches are the Motivational Interviewing Skill Code (MISC) [<xref ref-type="bibr" rid="ref7">7</xref>] and the Motivational Interviewing Treatment Integrity (MITI) framework [<xref ref-type="bibr" rid="ref8">8</xref>]. The MISC is a widely used coding system that classifies every facilitator and participant interaction within a session, allowing the analysis of conversation elements such as change talk and sustain talk [<xref ref-type="bibr" rid="ref7">7</xref>]. The MITI framework was developed as a more practical fidelity tool, focusing on a smaller set of global competencies (eg, empathy, MI spirit) and behaviors (eg, reflections, questions, affirmations). These methods provide a structured way to monitor facilitator proficiency and offer targeted feedback in training and supervision contexts. A key limitation of these methods, however, is their reliance on time-consuming processes such as verbatim transcription and manual coding of facilitator-participant interactions. This dependence significantly limits their scalability and feasibility in routine training and implementation settings. Consequently, the use of these methods to evaluate facilitators&#x2019; implementation of MI techniques is often impractical.</p><p>Prior efforts to automate MI fidelity coding have applied a range of natural language processing (NLP) and text classification methods to this problem, including rule-based systems, bag-of-words models, recurrent neural networks [<xref ref-type="bibr" rid="ref9">9</xref>], and transformer-based classifiers trained on annotated MI corpora [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. These approaches have shown that automated coding of facilitator utterances is feasible and can approximate human-level reliability for selected MI techniques. However, they typically require large, labeled training datasets, domain-specific model fine-tuning, and substantial computational infrastructure, resources that are rarely available in real-world settings. Moreover, most of this work has been conducted on English-language clinical therapy sessions with adult populations, limiting its applicability to other intervention contexts.</p><p>Recent advances in generative AI, specifically large language models (LLMs) [<xref ref-type="bibr" rid="ref12">12</xref>], offer a qualitatively different approach [<xref ref-type="bibr" rid="ref13">13</xref>]. Unlike earlier NLP methods, LLMs can perform text classification tasks through in-context learning, that is, by following natural-language instructions in a prompt [<xref ref-type="bibr" rid="ref14">14</xref>], without requiring labeled training data or model fine-tuning [<xref ref-type="bibr" rid="ref13">13</xref>]. This capability dramatically lowers the technical barrier to entry and makes automated coding accessible to research teams that lack a specialized machine learning infrastructure. Early evidence suggests that LLMs can achieve competitive performance on psychotherapy coding tasks, including the detection of therapeutic techniques in session transcripts [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. However, evidence remains limited on whether LLMs can identify MI techniques in non-English standardized practice sessions for preventive interventions designed for adolescents in school-based settings.</p><p>The present study addresses this gap by examining the feasibility of using an LLM to assess MI implementation within the context of a brief substance use preventive intervention for adolescents delivered in Spanish. Using transcripts from Spanish-language facilitator practice sessions for a school-based preventive intervention, in which a chatbot simulated adolescent responses, we evaluated whether an AI-based coder can reliably detect the use of MI techniques (specifically OARS) by comparing its recognition of MI techniques against those of trained human coders. Conceptually, these 2 coding tasks map onto distinct dimensions of implementation fidelity as described in the framework of Carroll et al [<xref ref-type="bibr" rid="ref17">17</xref>]: quality of delivery, operationalized as the appropriate use of MI techniques at the facilitator-message level, and adherence, operationalized through a 21-item checklist capturing the prescribed stages of a brief MI-based intervention. By examining both dimensions, the current study provides a more comprehensive test of the feasibility of AI-assisted fidelity measurement than studies targeting a single dimension. We examined (1) intercoder agreement between human coders and an AI-based coder and (2) whether the AI coder differed systematically from human coders in its ability to recognize specific MI techniques and implementation adherence. We hypothesized that the AI-based coder would demonstrate moderate-to-substantial agreement with human coders across (1) commonly used MI techniques and (2) indicators of MI implementation adherence. We further hypothesized that the AI coder would not exhibit systematic bias relative to human coders.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>Facilitator messages from session transcripts were coded independently by trained human coders and an AI-based coder using a common MI coding frame. Agreement was examined at two levels: (1) at the level of facilitator messages for the identification of specific MI techniques and (2) at the session level, for assessing implementation adherence across intervention stages. Reporting follows the GRRAS (Guidelines for Reporting Reliability and Agreement Studies) for the intercoder agreement components [<xref ref-type="bibr" rid="ref18">18</xref>] and is informed by the TRIPOD-LLM (Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis&#x2013;Large Language Models) reporting guidance for studies that develop or evaluate LLMs [<xref ref-type="bibr" rid="ref19">19</xref>]; completed checklists for both guidelines are provided <xref ref-type="supplementary-material" rid="app1">Checklists 1</xref> and <xref ref-type="supplementary-material" rid="app2">2</xref>.</p></sec><sec id="s2-2"><title>Intervention Context</title><p>The Brief Intervention Based on Motivational Interviewing program (<italic>Intervenci&#x00F3;n Breve Basada en Entrevista Motivacional</italic> [IBEM]) served as the exemplar intervention for this study. IBEM is a brief, school-based preventive intervention, developed in Colombia, and grounded in the principles of MI. Its goal is to support adolescents in reflecting on their motivations and behaviors related to substance use, with the aim of promoting healthier decision-making and reducing risk-related behaviors. The intervention consists of a single core session followed by 2 follow-up meetings. The initial session, lasting approximately 15 to 20 minutes, centers on a personalized MI-guided conversation about alcohol and other substance use. Facilitators conduct a brief assessment; provide individualized feedback; and guide adolescents in exploring their motivations, perceived risks, and readiness for change. Through this dialogue, adolescents identify self-directed goals and develop strategies to reduce or prevent substance use. The 2 follow-up sessions are conducted at 3 and 6 months and focus on reviewing progress toward previously established goals, discussing achievements, and identifying barriers and facilitators that may influence behavior change. IBEM has been developed in Colombia and was implemented also in Mexico and Brazil. Evaluations of IBEM in Colombia have shown the intervention to be efficacious in reducing adolescent alcohol use and its associated risks [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. In 2021, IBEM received the National Award for Best Practices in Prevention, granted by the Colombian Ministry of Health and Social Protection in recognition of its contribution to prevention practice.</p></sec><sec id="s2-3"><title>Participants</title><p>Two expert human coders were recruited in consultation with IBEM program developers [<xref ref-type="bibr" rid="ref22">22</xref>] and were invited to participate in the study based on their prior certification in the intervention and demonstrated experience implementing it in school-based settings. Both coders were Colombian professionals and native Spanish speakers, reflecting the linguistic and cultural context in which IBEM was originally developed and delivered. Each coder held a professional degree in psychology and postgraduate-level training and completed formal instruction in MI. IBEM certification required a 32-hour training in adolescent substance use (ie, alcohol, tobacco, and other illicit drug use), MI principles and techniques, role-playing, and implementation evaluation. Both coders possessed over 5 years of experience delivering IBEM in real-world implementation contexts.</p></sec><sec id="s2-4"><title>IBEM Transcripts</title><p>Two transcripts were selected from a pool of 10 IBEM sessions conducted in Spanish by certified implementers, originally collected as part of a previous research project focused on developing an AI chatbot to simulate adolescent responses during IBEM delivery. IBEM sessions, therefore, were not routine clinical encounters; certified expert facilitators delivered complete IBEM sessions while interacting with a simulated adolescent, which allowed facilitators to exercise the full MI protocol under controlled conditions. The selection of IBEM sessions was based on three considerations: (1) expert review indicated that the sessions reflected typical IBEM delivery rather than either flawless delivery or substantial omission of protocol components, (2) sessions were of average length relative to the pool of sessions, and (3) coders were not assigned to transcripts from sessions they had personally facilitated. A facilitator message was defined as a single, uninterrupted facilitator turn in the transcript; when a turn contained several sentences, it was treated as 1 message and could receive multiple technique codes. The selected transcripts yielded 72 facilitator messages (32 messages from session 1 and 40 messages from session 2).</p></sec><sec id="s2-5"><title>Coding Framework</title><sec id="s2-5-1"><title>MI Techniques</title><p>The coding scheme was adapted from the MISC and the MITI frameworks and organized around the OARS. Consistent with these systems, facilitator behaviors were operationalized as discrete, message-level codes rather than global session ratings; higher-order MITI global dimensions (eg, empathy, MI spirit) were intentionally beyond the scope of this feasibility study. Each facilitator message was classified, independently by each coder, into one of the following categories: (1) reflections, (2) questioning, (3) confrontational, (4) affirmative, (5) giving information, and (6) not categorized. <italic>Reflections</italic> were coded when the facilitator repeated or paraphrased adolescent statements, with <italic>simple reflections</italic> mirroring content literally and <italic>complex reflections</italic> capturing underlying meaning or emotion. <italic>Questioning</italic> was distinguished as <italic>open</italic> (inviting elaboration) or <italic>closed</italic> (eliciting brief, yes or no responses). <italic>Confrontational</italic> included direct disagreement or corrective statements. <italic>Affirmative</italic> highlighted participant strengths or efforts. <italic>Giving information</italic> involved offering factual, neutral content related to alcohol, tobacco, or other illicit drug use. <italic>Not categorized</italic> was used for messages such as greetings or farewells that did not fall into the other categories. Coders were allowed to assign more than one technique to the same message. A provisional &#x201C;unclear&#x201D; category was initially available for messages whose function was ambiguous but was not endorsed by any coder during piloting and was dropped from the final scheme; residual ambiguous messages were coded as not categorized.</p></sec><sec id="s2-5-2"><title>Implementation Adherence</title><p>Coders documented the <italic>presence</italic> (coded 1) or <italic>absence</italic> (coded 0) of IBEM intervention steps using a 21-item intervention adherence checklist that covered the 6 stages of the IBEM implementation protocol. In stage 1, Presentation and Contextualization assessed whether the facilitator introduced themselves, explained the program&#x2019;s objectives, clarified voluntariness and confidentiality, and collected demographic data. Stage 2, Risk Assessment, captured whether facilitators reviewed student responses to the IBEM brief assessment surveys, explored recent behavior related to alcohol use in greater detail, communicated the risk level, and postponed notification of risk level until the end of the session when risk was severe. Stage 3, Evocation and Motivators, evaluated whether facilitators explored the student&#x2019;s knowledge about alcohol, tobacco, and drugs, elicited personal motivators (eg, favorite activities), and linked these motivators to information about risks and consequences. Stage 4, Importance and Confidence, assessed whether the facilitator explained the scales, inquired about the reasons for scores, and explored the student&#x2019;s confidence in their ability to change or maintain behaviors. Stage 5, Goal setting and Strategies, captured whether a concrete action plan and start date were established, and whether goals and strategies were generated by the student rather than the facilitator. Stage 6, Summary and Closure, evaluated whether the facilitator summarized key points of the session including risk level, motivators, effects, goals, and strategies, and provided information about the timing of the next session. Each of the 21 items was scored dichotomously at the session level, indicating whether the corresponding protocol component was <italic>present</italic> (coded 1) or <italic>absent</italic> (coded 0) in the session. The implementation-adherence score for a session was therefore the number of endorsed items out of 21, and agreement analyses were conducted across the 42 item-by-session ratings (21 items &#x00D7; 2 sessions) for each coder.</p></sec><sec id="s2-5-3"><title>AI-Based Coding</title><p>An AI-based coder was developed using OpenAI&#x2019;s GPT 4.1 model [<xref ref-type="bibr" rid="ref23">23</xref>], accessed through the API. Facilitator messages were submitted to GPT 4.1 along with a structured prompt derived from the coding manuals. The prompt was designed to replicate the decision rules used by the human coders and included (1) definitions of each MI technique category (eg, reflections, questions, affirmations), (2) explicit instructions for distinguishing between similar categories (eg, open vs closed questions, simple vs complex reflections), and (3) guidance on how to handle ambiguous or irrelevant facilitator statements (eg, greetings or logistical remarks not otherwise categorized). The prompt also emphasized that multiple MI technique categories could be flagged simultaneously. To increase consistency in the AI responses, we instructed the model to return its classifications in a structured data format known as JSON. JSON is a simple way of representing information as a list of labels and values that computers can read. For each message, the model produced a JSON entry indicating whether each message belonged to an IBEM implementation protocol step (<italic>TRUE</italic> or <italic>FALSE</italic>) and included a short explanation for its decision. Classification outputs were then checked using the <italic>Pydantic framework</italic> [<xref ref-type="bibr" rid="ref24">24</xref>], which validated whether the format of the AI&#x2019;s responses followed the expected JSON structure. In practice, this meant verifying that every AI-generated response included the required pieces of information (ie, TRUE/FALSE decisions and the short explanation). The pipeline was implemented in <italic>Python 3.12</italic> [<xref ref-type="bibr" rid="ref25">25</xref>], <italic>Pandas</italic> for data management [<xref ref-type="bibr" rid="ref26">26</xref>], <italic>PyArrow</italic> for parquet handling [<xref ref-type="bibr" rid="ref27">27</xref>], <italic>Pydantic</italic> for schema validation, and the official <italic>OpenAI</italic> module for API access [<xref ref-type="bibr" rid="ref28">28</xref>]. In brief, JSON provided a consistent, machine-readable structure for each classification, pairing every category decision with a short textual rationale, and Pydantic enforced this schema so that any malformed or incomplete model output (eg, a missing decision field or an invalid category value) was automatically flagged and corrected before the AI codes were merged with the human-coded data set for analysis. This step was intended to improve the reproducibility and auditability of the automated coding rather than to alter the message classifications.</p></sec></sec><sec id="s2-6"><title>Procedures</title><p>Human coders completed a structured virtual training based on the MI coding manual prior to transcript review. Coding involved 2 distinct tasks. First, coders evaluated MI techniques at the facilitator-message level, assigning codes to each individual facilitator message. Second, coders assessed intervention adherence at the item by session level using the 21-item checklist of required stages of IBEM implementation. After independently coding the first transcript, each human coder met individually with the research team to clarify ambiguities in the coding process. Minor refinements to category definitions were incorporated before the coding of the second transcript began. Coders were blinded to one another&#x2019;s ratings and the AI-generated codes throughout the process. AI codes were generated in parallel and merged with the human-coded dataset for comparative analyses.</p></sec><sec id="s2-7"><title>Statistical Analysis</title><p>Intercoder agreement for MI techniques was assessed using Cohen &#x03BA; [<xref ref-type="bibr" rid="ref29">29</xref>], calculated separately for each technique category and pair of coders (coder 1 vs coder 2, coder 1 vs AI coder, and coder 2 vs AI coder). To examine whether coders differed systematically in their likelihood of identifying techniques for the same messages, Cochran <italic>Q</italic> tests were conducted for each technique across all 3 coders. Descriptive statistics were used to summarize implementation adherence ratings by item and session (ie, 21 items &#x00D7; 2 sessions = 42 ratings for each coder). Overall intercoder agreement among the 3 coders was estimated using Fleiss &#x03BA; [<xref ref-type="bibr" rid="ref30">30</xref>]. Pairwise intercoder agreement (coder 1 vs coder 2, coder 1 vs AI coder, and coder 2 vs AI coder) was evaluated using Cohen &#x03BA; with 95% CIs. In addition, percentage agreement between pairs of coders was calculated to facilitate the interpretation of intercoder agreement statistics. Following standard practice, &#x03BA; values were interpreted using the benchmarks proposed by Landis and Koch [<xref ref-type="bibr" rid="ref31">31</xref>], specified a priori, in which values of 0.00 to 0.20 indicate slight agreement, 0.21 to 0.40 fair, 0.41 to 0.60 moderate, 0.61 to 0.80 substantial, and 0.81 to 1.00 almost perfect agreement.</p></sec><sec id="s2-8"><title>Ethical Considerations</title><p>All procedures involving human participants were reviewed and approved by the University of Miami Institutional Review Board (ID: 20250114; approval date: March 7, 2025). The study was conducted in accordance with the ethical standards of the institutional research committee. Informed consent was obtained through 2 separate processes: facilitators who participated in the original IBEM sessions provided consent to share the transcripts of their sessions. Prior to any analysis, all transcripts were deidentified by removing names and other direct identifiers so that messages submitted for coding did not contain personally identifying information. Deidentified facilitator messages were transmitted to OpenAI&#x2019;s GPT 4.1 through the API solely for classification; under the API terms in effect at the time of the study, submitted data were not used to train or improve OpenAI models.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Identification of MI Techniques</title><p>Across the 72 facilitator messages, <italic>open questions</italic> represented the most frequently coded MI technique across the 2 human coders consisting of 29% (n=21) and 31% (n=22) of the messages, respectively. <italic>Giving information</italic> was identified in 19% (n=14) of the messages by both coders. <italic>Affirmations</italic> were coded in 18% (n=13) and 15% (n=11) of the messages, and <italic>closed questions</italic> were coded in 17% (n=12) and 18% (n=13) of the messages, respectively. <italic>Reflective</italic> statements occurred less frequently; <italic>simple reflections</italic> were identified in 6% (n=4) and 4% (n=3) of the messages, and <italic>complex reflections</italic> were identified in 6% (n=4) and 4% (n=3) of the messages, respectively. <italic>Confrontation</italic> was identified in one (1%) instance by coder 2 and not by coder 1. Messages <italic>not otherwise categorized</italic> accounted for 18% (n=13) and 21% (n=15) of the messages. The AI-based coder demonstrated a generally similar coding distribution of facilitator messages. <italic>Open questions</italic> were identified in 28% (n=20) of the messages, followed by <italic>closed questions</italic> (n=16, 22%), <italic>affirmations</italic> (n=14, 19%), and <italic>giving information</italic> (n=14, 19%). <italic>Simple reflections</italic> were identified in 3% (n=2) of the messages, and no <italic>complex reflections</italic> were detected. Messages coded as not categorized comprised 17% (n=12) of the total messages.</p><p>Cochran <italic>Q</italic> tests revealed no statistically significant differences in the percentages of MI technique use across the 3 coders for any category: <italic>affirmations</italic>, Q(2, n=72)=1.56, <italic>P</italic>=.46; <italic>not categorized</italic>, Q(2, n=72)=1.56, <italic>P</italic>=.46; <italic>open questions</italic>, Q(2, n=72)=0.43, <italic>P</italic>=.81; <italic>closed questions</italic>, Q(2, n=72)=1.73, <italic>P</italic>=.42; <italic>giving information</italic>, Q(2, n=72)=0.00, <italic>P</italic>&#x003E;.99; <italic>simple reflections</italic>, Q(2, n=72)=1.20, <italic>P</italic>=.55; <italic>confrontation</italic>, Q(2, n=72)=2.00, <italic>P</italic>=.37; and <italic>complex reflections</italic>, Q(2, n=72)=5.20, <italic>P</italic>=.07. The unclear category was not evaluated because it was not endorsed by any coder. The distributions of MI techniques by coder and session are presented in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Facilitator message coding frequencies for motivational interviewing (MI) techniques by coder and session.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95964_fig01.png"/></fig></sec><sec id="s3-2"><title>Intercoder Agreement</title><p>Between the 2 human coders, agreement was high for <italic>giving information</italic> (&#x03BA;=0.91) and substantial for <italic>affirmations</italic>, <italic>open questions, closed questions</italic>, and messages <italic>not categorized</italic> (&#x03BA; values ranging from 0.66 to 0.82). Agreement was lower for <italic>complex reflections</italic> (&#x03BA;=0.55) and lowest for <italic>simple reflections</italic> (&#x03BA;=0.25). The AI-based coder showed a similar pattern of agreement with both human coders. For the most frequently coded categories (<italic>giving information, affirmations,</italic> and <italic>open questions</italic>), AI-human agreement was substantial to high (&#x03BA; =0.66-0.91). The main discrepancies involved low-frequency categories: the AI coder did not identify any complex reflections, resulting in &#x03BA; of 0.00 for both AI-human pairs, and agreement on simple reflections was inconsistent across pairs (&#x03BA;=0.31 with coder 1 vs &#x03BA;=0.79 with coder 2). Agreement on closed questions was moderate for the AI-human pairs (&#x03BA;=0.47 and 0.53), compared with substantial agreement between the two human coders (&#x03BA;=0.66). Cohen &#x03BA; coefficients with 95% CIs for all coder pairs are presented in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Cohen &#x03BA; coefficients and pairwise percentage agreement across pairs of coders<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Category</td><td align="left" valign="bottom" colspan="2">Coder 1 vs coder 2</td><td align="left" valign="bottom" colspan="2">Coder 1 vs AI coder</td><td align="left" valign="bottom" colspan="2">Coder 2 vs AI coder</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">&#x03BA; (95% CI)</td><td align="left" valign="bottom">Percentage agreement</td><td align="left" valign="bottom">&#x03BA; (95% CI)</td><td align="left" valign="bottom">Percentage agreement</td><td align="left" valign="bottom">&#x03BA; (95% CI)</td><td align="left" valign="bottom">Percentage agreement</td></tr></thead><tbody><tr><td align="left" valign="top">Affirmations</td><td align="left" valign="top">0.70 (0.48 to 0.92)</td><td align="left" valign="top">91.7</td><td align="left" valign="top">0.77 (0.58 to 0.96)</td><td align="left" valign="top">93.1</td><td align="left" valign="top">0.66 (0.43 to 0.89)</td><td align="left" valign="top">90.3</td></tr><tr><td align="left" valign="top">Confrontations</td><td align="left" valign="top">0.00 (0.00 to 0.00)</td><td align="left" valign="top">98.6</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">100</td><td align="left" valign="top">0.00 (0.00 to 0.00)</td><td align="left" valign="top">98.6</td></tr><tr><td align="left" valign="top">Closed questions</td><td align="left" valign="top">0.66 (0.43 to 0.89)</td><td align="left" valign="top">90.3</td><td align="left" valign="top">0.47 (0.22 to 0.72)</td><td align="left" valign="top">83.3</td><td align="left" valign="top">0.53 (0.28 to 0.77)</td><td align="left" valign="top">84.7</td></tr><tr><td align="left" valign="top">Complex reflections</td><td align="left" valign="top">0.55 (0.10 to 1.00)</td><td align="left" valign="top">95.8</td><td align="left" valign="top">0.00 (&#x2212;0.00 to 0.00)</td><td align="left" valign="top">94.4</td><td align="left" valign="top">0.00 (&#x2212;0.00 to 0.00)</td><td align="left" valign="top">95.8</td></tr><tr><td align="left" valign="top">Not categorized</td><td align="left" valign="top">0.82 (0.66 to 0.99)</td><td align="left" valign="top">94.4</td><td align="left" valign="top">0.66 (0.43 to 0.89)</td><td align="left" valign="top">90.3</td><td align="left" valign="top">0.68 (0.46 to 0.90)</td><td align="left" valign="top">90.3</td></tr><tr><td align="left" valign="top">Open questions</td><td align="left" valign="top">0.70 (0.52 to 0.88)</td><td align="left" valign="top">87.5</td><td align="left" valign="top">0.69 (0.51 to 0.88)</td><td align="left" valign="top">87.5</td><td align="left" valign="top">0.66 (0.47 to 0.85)</td><td align="left" valign="top">86.1</td></tr><tr><td align="left" valign="top">Giving information</td><td align="left" valign="top">0.91 (0.79 to 1.00)</td><td align="left" valign="top">97.2</td><td align="left" valign="top">0.91 (0.79 to 1.00)</td><td align="left" valign="top">97.2</td><td align="left" valign="top">0.91 (0.79 to 1.00)</td><td align="left" valign="top">97.2</td></tr><tr><td align="left" valign="top">Simple reflections</td><td align="left" valign="top">0.25 (&#x2212;0.20 to 0.70)</td><td align="left" valign="top">93.1</td><td align="left" valign="top">0.31 (&#x2212;0.19 to 0.80)</td><td align="left" valign="top">94.4</td><td align="left" valign="top">0.79 (0.40 to 1.00)</td><td align="left" valign="top">98.6</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Interpretive thresholds for &#x03BA; are as follows: &#x003C;0.00=poor; 0.00&#x2013;0.20=slight; 0.21&#x2013;0.40=fair; 0.41&#x2013;0.60=moderate; 0.61&#x2013;0.80=substantial; and 0.81&#x2013;1.00=almost perfect.</p></fn><fn id="table1fn2"><p><sup>b</sup>This category did not occur and &#x03BA; could not be estimated. </p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>IBEM Implementation Adherence</title><p>Across the 2 IBEM sessions, adherence to implementation stages varied by session but showed substantial overlap among AI and human coders. Using the 21-item checklist, coders 1 and 2 endorsed 12 (57.1%) and 10 (47.6%) items in session 1, respectively, whereas the AI coder endorsed 11 (52.4%). In session 2, endorsement rates were higher for all coders (coder 1: n=16, 76.2%; coder 2: n=17, 81.0%; AI coder: n=20, 95.2%). When checklist ratings were pooled between sessions (42 ratings), full 3-way agreement occurred for 28 (66.7%) of the items. In terms of pairwise percentage agreement, concordance was 85.7% between the AI coder and coder 2, 78.6% between the 2 human coders, and 69.0% between the AI coder and coder 1. Disagreements clustered primarily in stage 2 (Risk Assessment) and stage 4 (Importance and Confidence), whereas stage 1 (Presentation and Contextualization) showed near-uniform agreement across coders. Overall, interrater agreement across the 3 coders, assessed with Fleiss &#x03BA;, was moderate (&#x03BA;=0.487, <italic>z</italic>=5.47; <italic>P</italic>&#x003C;.001). Pairwise Cohen &#x03BA; indicated moderate agreement between the 2 human coders (&#x03BA;=0.526, 95% CI 0.256-0.797) and substantial agreement between coder 2 and the AI coder (&#x03BA;=0.669, 95% CI 0.432-0.907), whereas agreement between coder 1 and the AI coder was fair (&#x03BA;=0.264, 95% CI 0.043-0.571). <xref ref-type="fig" rid="figure2">Figure 2</xref> presents agreement patterns for each adherence item by the stage of IBEM implementation across the 2 IBEM sessions.</p><p>For each of the 21 checklist items in each of the 2 sessions, &#x201C;agreement&#x201D; denotes concordance among coders on the same present (1) or absent (0) rating. The figure displays both 3-way agreement (all 3 coders assigning the same rating) and pairwise agreement, organized by the 6 IBEM implementation stages.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Agreement among coders (human and AI) for Intervenci&#x00F3;n Breve Basada en Entrevista Motivacional (IBEM) implementation adherence items by stage.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e95964_fig02.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study examined the use of an LLM to automate the identification of intervention techniques and implementation adherence coding within the context of a brief MI-based preventive intervention for adolescents. Agreement between the AI-based coder and trained human coders was assessed at 2 levels: individual facilitator messages for MI technique identification and item-level checklists for session-specific implementation adherence. Using the framework of Carroll et al [<xref ref-type="bibr" rid="ref17">17</xref>], these may be thought of as implementation quality of delivery and adherence, 2 core components of implementation fidelity. Proctor et al [<xref ref-type="bibr" rid="ref32">32</xref>] position fidelity as a key implementation outcome that mediates the relationship between implementation strategies and intervention effectiveness, underscoring the importance of scalable fidelity measurement. Overall, our findings provide support for the feasibility of AI-assisted coding of MI techniques and implementation adherence, although they reveal specific limitations that require further investigation. These results were derived from a proof-of-concept dataset and were therefore intended to establish initial feasibility rather than to justify large-scale deployment or to guide consequential coding decisions in practice. Accordingly, the estimates reported below should be interpreted with caution.</p><p>Consistent with our first hypothesis, the AI-based coder demonstrated moderate-to-substantial agreement with human coders across the most commonly used MI techniques. Agreement was particularly strong for <italic>giving information</italic>, <italic>affirmations</italic>, and <italic>open questions</italic>, techniques characterized by relatively explicit linguistic markers that may be easier for automated detection [<xref ref-type="bibr" rid="ref10">10</xref>]. Importantly, Cochran <italic>Q</italic> tests revealed no statistically significant differences in the percentages of MI technique use across the 3 coders for any technique category, supporting our hypothesis that the AI system would not exhibit systematic bias in category assignment. These agreement levels are broadly consistent with those reported in prior automated MI coding research. For example, Tanana et al [<xref ref-type="bibr" rid="ref9">9</xref>] found that both deep-learning models achieved high agreement with human coders for open and closed questions, giving information, and affirmations (all &#x03BA;&#x003E;.50), with modestly lower agreement for simple and complex reflections (&#x03BA; values between 0.30 and 0.50). The present study, using a general-purpose LLM without task-specific training, achieved comparable or higher agreement for these high-frequency categories, suggesting that advances in language modeling may reduce the need for domain-specific model training.</p><p>We note, however, that the AI coder&#x2019;s performance on <italic>reflection</italic> statements was notably weaker than other MI techniques. The AI system failed to identify any <italic>complex reflections</italic>, producing &#x03BA; values of 0.00 for both human-AI comparisons. <italic>Complex reflections</italic> require inferring implicit meaning, emotional undertones, and rephrasing that extends beyond the literal content of participant statements, processes that demand a level of pragmatic interpretation that current LLMs struggle with [<xref ref-type="bibr" rid="ref13">13</xref>]. Notably, <italic>complex reflections</italic> were also among the least frequently coded categories by human raters, and agreement between the 2 human coders on this category was only moderate (&#x03BA;=0.55), suggesting that these distinctions are inherently difficult even for trained experts. For simple reflections, an interesting asymmetry emerged: agreement between the AI and coder 2 was substantial (&#x03BA;=0.79), whereas agreement with coder 1 was fair (&#x03BA;=0.31). This variability likely reflects individual differences in how coders operationalize the boundary between <italic>simple reflections</italic> and similar categories such as <italic>closed questions</italic> or paraphrases, a well-documented challenge in MI coding [<xref ref-type="bibr" rid="ref33">33</xref>], and suggests that coding MI techniques is a difficult task even for human experts. The AI coder&#x2019;s inability to detect complex reflections echoes findings from earlier NLP-based approaches. Tanana et al [<xref ref-type="bibr" rid="ref9">9</xref>] similarly reported lower accuracy for reflections compared with questions, and Atkins et al [<xref ref-type="bibr" rid="ref10">10</xref>] noted that categories requiring pragmatic inference have the greatest challenge for statistical classifiers. The present findings suggest that this limitation persists even with more advanced language models.</p><p>The results from the IBEM 21-item implementation adherence checklist were similarly promising. Full 3-way agreement occurred for two-thirds of the checklist items (n=28, 66.7%), and the AI coder achieved the highest pairwise percentage agreement with coder 2 (n=36, 85.7%). Pairwise &#x03BA; between the AI coder and coder 2 was substantial (&#x03BA;=0.66), exceeding the moderate agreement observed between the 2 human coders themselves (&#x03BA;=0.52). These findings suggest that the AI coder was able to interpret session structure and procedural adherence with a degree of reliability comparable to human-to-human agreement. Checklists are commonly used in implementation settings to monitor facilitator compliance with program protocols [<xref ref-type="bibr" rid="ref34">34</xref>], and the present results indicate that AI-based monitoring could meaningfully support this function and speed up the pace of adherence monitoring.</p><p>That said, agreement between the AI coder and Coder 1 was only fair (&#x03BA;=0.264), indicating unreliable concordance for this pair. This discrepancy, however, should be interpreted in light of the disagreement among the human coders themselves rather than as evidence of a problem unique to the AI coder. The 2 human coders agreed only moderately on adherence (&#x03BA;=0.52), so there was no single human &#x201C;gold standard&#x201D; against which the AI coder could be benchmarked. Critically, the AI coder aligned substantially with coder 2 (&#x03BA;=0.66) but only fairly with coder 1, and coder 1 also diverged from coder 2 to a comparable degree. Coder 1 was therefore the most distinct rater of the three, and the AI coder&#x2019;s weaker agreement with that coder largely mirrors a preexisting difference between the 2 humans.</p><p>Disagreements clustered in stage 2 (Risk Assessment) and stage 4 (Importance and Confidence), suggesting that adherence checklist items requiring interpretation of facilitator intent, rather than the presence of explicit content, remain challenging for both human and AI coders. It is also worth noting that checklist-based adherence assessment has inherent limitations: sessions are highly sensitive to the context of each interaction. For instance, some adolescents may be more engaged or collaborative, while others may be more reserved, leading facilitators to adapt their delivery accordingly. These contextual variations may produce natural differences in how facilitators approach each stage, which in turn complicates the interpretation of adherence ratings regardless of whether the coder is human or AI.</p><p>From an implementation science perspective, the AI coder&#x2019;s ability to consistently recognize most core MI techniques and procedural steps represents a meaningful advancement for scalable fidelity monitoring. Returning to the conceptualization of implementation fidelity of Carroll et al [<xref ref-type="bibr" rid="ref17">17</xref>], the AI coder performed well on adherence measurement, where agreement with one human coder exceeded human-to-human agreement, and on identification of MI techniques (a potential indicator of quality of intervention delivery) for categories with explicit linguistic markers. This pattern suggests that LLM-based coding may be most immediately viable for the adherence dimension of fidelity, which relies on detecting the presence or absence of prescribed intervention components, while the quality of delivery dimension of fidelity requires further development for higher-inference techniques such as complex reflections. Manual coding of MI sessions is time-intensive and costly, and its reliance on trained human coders limits the frequency of fidelity monitoring. An AI-assisted approach could reduce human workload, increase the feasibility of conducting routine fidelity checks, and enable more timely feedback to facilitators, particularly in resource-constrained or geographically dispersed implementation settings. In practical terms, these results point toward the need for a human-in-the-loop workflow rather than full automation. In such a model, the AI coder would &#x201C;precode&#x201D; facilitator messages and adherence items and flag low-confidence or higher-inference cases, such as complex reflections and the stage 2 and stage 4 items where agreement was the weakest.</p><p>Finally, a human-in-the-loop workflow should treat the AI coder as an annotation instrument and assess whether its outputs systematically change the distribution of codes across categories. For example, Xu et al [<xref ref-type="bibr" rid="ref35">35</xref>] found that an LLM annotator shifted the overall distribution of emotional-tone labels toward the &#x201C;very negative&#x201D; category. In our study, the AI coder identified no complex reflections, whereas the human coders identified some examples of this higher-inference technique. Although these counts are too small to establish systematic undercoding, they illustrate why human oversight should examine whether the AI consistently over- or underrepresents specific categories before its codes are used in fidelity monitoring or as training labels.</p></sec><sec id="s4-2"><title>Limitations</title><p>Several limitations should be considered when interpreting these findings. First, the study analyzed only 2 session transcripts comprising 72 facilitator messages and 42 adherence checklist items, which limits the generalizability of the study findings. With the small number of observations per category, the agreement estimates have limited precision, as reflected in the wide CIs, and the results should be regarded as preliminary evidence. Second, only 2 human coders participated, which restricts the estimation of human-to-human reliability and does not capture the full range of variability that would be observed in a larger coder panel. Because human-to-human reliability was itself moderate, comparisons between the AI coder and each individual coder were sensitive to that coder&#x2019;s idiosyncratic coding style. A larger and more diverse panel of coders would provide a more stable human reference standard against which to benchmark the AI coder. Third, the study evaluated a single LLM (GPT 4.1) with a single prompt configuration; performance may differ across models, prompt versions, and prompt engineering strategies, and the study results should not be generalized to other LLMs without direct comparison. Fourth, all transcripts were in Spanish from a Colombian intervention context, and it remains unknown whether agreement levels would hold for MI sessions conducted in other languages or cultural settings. Relatedly, the transcripts were drawn from standardized facilitator practice sessions in which the interlocutor was a chatbot simulating adolescent responses rather than a live client; although this design isolated facilitator MI behaviors under controlled conditions, conversational dynamics may differ from naturalistic school-based sessions, and the findings should be replicated with live adolescent interactions before broader use. Finally, the coding framework used in this study focused on behavioral counts of MI techniques and binary adherence checklist items; the study did not evaluate higher-order global competencies such as empathy, which may present additional challenges for automated classification.</p><p>Future research should address these limitations by (1) expanding the number and diversity of transcripts to include sessions from multiple intervention sites and facilitators, (2) comparing the performance of multiple LLMs and systematically evaluating the impact of prompt engineering on classification accuracy, (3) examining whether AI-assisted feedback can improve both proximal (eg, facilitator adherence) and distal (eg, program sustainability) implementation outcomes in prospective designs, and (4) extending the coding framework to include global MI competencies (eg, empathy) to test automated coding of complex MI techniques. Additionally, cross-linguistic validation studies are needed to determine whether the present findings generalize to MI sessions conducted in English and other languages. Integrating AI-based coding into real-time supervision workflows represents a particularly promising direction for translating these findings into implementation practice.</p></sec></sec></body><back><ack><p>The authors thank all collaborators and participants who contributed to this study and supported the implementation of the intervention.</p><p>During manuscript preparation, the authors used Copilot, a generative AI tool, to support grammar review and text editing. All AI-assisted output was reviewed and revised by the authors, who take full responsibility for the content of this manuscript. Separately, OpenAI&#x2019;s GPT 4.1 was used exclusively as a study instrument for the automated coding analyses described in the <italic>Methods</italic> section and not to generate or draft any portion of the manuscript text.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the IMPACT Lab and by the National Institutes of Health under awards R01MH120601, R01DA059527, R01MH131207, R01DA054885, and R01MH128878. The funders had no role in the study design, data collection, analysis, interpretation of results, or writing of the manuscript.</p></sec><sec><title>Data Availability</title><p>The datasets analyzed during the current study are available from the corresponding author on reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>FC and ECB conceived the study. FC conducted the analyses and drafted the manuscript. ECB, JM-T, CC, SMSG, AP-G, and VG contributed to the study design and interpretation of the results. RB contributed to the development of the figures, interpretation of the data, and manuscript revision. ECB, SMSG, AP-G, and VG provided supervision and methodological guidance. All authors contributed to revising the manuscript for important intellectual content and approved the final version.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">GRRAS</term><def><p>Guidelines for Reporting Reliability and Agreement Studies</p></def></def-item><def-item><term id="abb2">IBEM</term><def><p>Intervenci&#x00F3;n Breve Basada en Entrevista Motivacional [Brief Intervention Based on Motivational Interviewing program]</p></def></def-item><def-item><term id="abb3">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb4">MI</term><def><p>motivational interviewing</p></def></def-item><def-item><term id="abb5">MISC</term><def><p>Motivational Interviewing Skill Code</p></def></def-item><def-item><term id="abb6">MITI</term><def><p>Motivational Interviewing Treatment Integrity</p></def></def-item><def-item><term id="abb7">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb8">OARS</term><def><p>open questions, affirmations, reflections, and summaries</p></def></def-item><def-item><term id="abb9">TRIPOD-LLM</term><def><p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis&#x2013;Large Language Models</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Miller</surname><given-names>WR</given-names> </name><name name-style="western"><surname>Rollnick</surname><given-names>S</given-names> </name></person-group><source>Motivational Interviewing: Helping People Change</source><year>2013</year><edition>3</edition><publisher-name>Guilford Press</publisher-name><pub-id pub-id-type="other">978-1-60918-227-4</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dean</surname><given-names>S</given-names> </name><name name-style="western"><surname>Britt</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bell</surname><given-names>E</given-names> </name><name name-style="western"><surname>Stanley</surname><given-names>J</given-names> </name><name name-style="western"><surname>Collings</surname><given-names>S</given-names> </name></person-group><article-title>Motivational interviewing to enhance adolescent mental health treatment engagement: a randomized clinical trial</article-title><source>Psychol Med</source><year>2016</year><month>07</month><volume>46</volume><issue>9</issue><fpage>1961</fpage><lpage>1969</lpage><pub-id pub-id-type="doi">10.1017/S0033291716000568</pub-id><pub-id pub-id-type="medline">27045520</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Melastuti</surname><given-names>E</given-names> </name><name name-style="western"><surname>Sukartini</surname><given-names>T</given-names> </name></person-group><article-title>Motivational interviewing as a problem solving intervention to improve adherence: review of the related literature</article-title><source>Indian J Public Health Res Dev</source><year>2019</year><volume>10</volume><issue>8</issue><fpage>2580</fpage><lpage>2584</lpage><pub-id pub-id-type="doi">10.5958/0976-5506.2019.02256.3</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>DiClemente</surname><given-names>CC</given-names> </name><name name-style="western"><surname>Corno</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Graydon</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Wiprovnick</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Knoblach</surname><given-names>DJ</given-names> </name></person-group><article-title>Motivational interviewing, enhancement, and brief interventions over the last decade: a review of reviews of efficacy and effectiveness</article-title><source>Psychol Addict Behav</source><year>2017</year><month>12</month><volume>31</volume><issue>8</issue><fpage>862</fpage><lpage>887</lpage><pub-id pub-id-type="doi">10.1037/adb0000318</pub-id><pub-id pub-id-type="medline">29199843</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hartzler</surname><given-names>B</given-names> </name><name name-style="western"><surname>Beadnell</surname><given-names>B</given-names> </name><name name-style="western"><surname>Rosengren</surname><given-names>DB</given-names> </name><name name-style="western"><surname>Dunn</surname><given-names>C</given-names> </name><name name-style="western"><surname>Baer</surname><given-names>JS</given-names> </name></person-group><article-title>Deconstructing proficiency in motivational interviewing: mechanics of skilful practitioner delivery during brief simulated encounters</article-title><source>Behav Cogn Psychother</source><year>2010</year><month>10</month><volume>38</volume><issue>5</issue><fpage>611</fpage><lpage>628</lpage><pub-id pub-id-type="doi">10.1017/S1352465810000329</pub-id><pub-id pub-id-type="medline">20615272</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schwalbe</surname><given-names>CS</given-names> </name><name name-style="western"><surname>Oh</surname><given-names>HY</given-names> </name><name name-style="western"><surname>Zweben</surname><given-names>A</given-names> </name></person-group><article-title>Sustaining motivational interviewing: a meta-analysis of training studies</article-title><source>Addiction</source><year>2014</year><month>08</month><volume>109</volume><issue>8</issue><fpage>1287</fpage><lpage>1294</lpage><pub-id pub-id-type="doi">10.1111/add.12558</pub-id><pub-id pub-id-type="medline">24661345</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moyers</surname><given-names>T</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Catley</surname><given-names>D</given-names> </name><name name-style="western"><surname>Harris</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Ahluwalia</surname><given-names>JS</given-names> </name></person-group><article-title>Assessing the integrity of motivational interviewing interventions: reliability of the motivational interviewing skills code</article-title><source>Behav Cogn Psychother</source><year>2003</year><volume>31</volume><issue>2</issue><fpage>177</fpage><lpage>184</lpage><pub-id pub-id-type="doi">10.1017/S1352465803002054</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moyers</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Rowell</surname><given-names>LN</given-names> </name><name name-style="western"><surname>Manuel</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Ernst</surname><given-names>D</given-names> </name><name name-style="western"><surname>Houck</surname><given-names>JM</given-names> </name></person-group><article-title>The Motivational Interviewing Treatment Integrity Code (MITI 4): rationale, preliminary reliability and validity</article-title><source>J Subst Abuse Treat</source><year>2016</year><month>06</month><volume>65</volume><fpage>36</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1016/j.jsat.2016.01.001</pub-id><pub-id pub-id-type="medline">26874558</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tanana</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hallgren</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Imel</surname><given-names>ZE</given-names> </name><name name-style="western"><surname>Atkins</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Srikumar</surname><given-names>V</given-names> </name></person-group><article-title>A comparison of natural language processing methods for automated coding of motivational interviewing</article-title><source>J Subst Abuse Treat</source><year>2016</year><month>06</month><volume>65</volume><fpage>43</fpage><lpage>50</lpage><pub-id pub-id-type="doi">10.1016/j.jsat.2016.01.006</pub-id><pub-id pub-id-type="medline">26944234</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Atkins</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Steyvers</surname><given-names>M</given-names> </name><name name-style="western"><surname>Imel</surname><given-names>ZE</given-names> </name><name name-style="western"><surname>Smyth</surname><given-names>P</given-names> </name></person-group><article-title>Scaling up the evaluation of psychotherapy: evaluating motivational interviewing fidelity via statistical text classification</article-title><source>Implementation Sci</source><year>2014</year><month>12</month><volume>9</volume><issue>1</issue><pub-id pub-id-type="doi">10.1186/1748-5908-9-49</pub-id><pub-id pub-id-type="medline">24758152</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Flemotomos</surname><given-names>N</given-names> </name><name name-style="western"><surname>Martinez</surname><given-names>VR</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Automated evaluation of psychotherapy skills using speech and language technologies</article-title><source>Behav Res Methods</source><year>2022</year><month>04</month><volume>54</volume><issue>2</issue><fpage>690</fpage><lpage>711</lpage><pub-id pub-id-type="doi">10.3758/s13428-021-01623-4</pub-id><pub-id pub-id-type="medline">34346043</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Vaswani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Parmar</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Attention is all you need</article-title><access-date>2026-08-12</access-date><conf-name>31st Conference on Neural Information Processing Systems (NIPS 2017)</conf-name><conf-date>Dec 5-7, 2017</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://papers.nips.cc/paper_files/paper/2017/hash/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html">https://papers.nips.cc/paper_files/paper/2017/hash/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html</ext-link></comment></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stade</surname><given-names>EC</given-names> </name><name name-style="western"><surname>Stirman</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Ungar</surname><given-names>LH</given-names> </name><etal/></person-group><article-title>Large language models could change the future of behavioral healthcare: a proposal for responsible development and evaluation</article-title><source>NPJ Mental Health Res</source><year>2024</year><volume>3</volume><issue>1</issue><fpage>12</fpage><pub-id pub-id-type="doi">10.1038/s44184-024-00056-z</pub-id><pub-id pub-id-type="medline">38609507</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kojima</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gu</surname><given-names>S (Shane</given-names> </name><name name-style="western"><surname>Reid</surname><given-names>M</given-names> </name><name name-style="western"><surname>Matsuo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Iwasawa</surname><given-names>Y</given-names> </name></person-group><article-title>Large language models are zero-shot reasoners</article-title><conf-name>Advances in Neural Information Processing Systems 35</conf-name><conf-date>Nov 28 to Dec 9, 2022</conf-date><pub-id pub-id-type="doi">10.52202/068431-1613</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ge</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><name name-style="western"><surname>Delk</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>JC</given-names> </name></person-group><article-title>A comparison of a large language model vs manual chart review for the extraction of data elements from the electronic health record</article-title><source>Gastroenterology</source><year>2024</year><month>04</month><volume>166</volume><issue>4</issue><fpage>707</fpage><lpage>709</lpage><pub-id pub-id-type="doi">10.1053/j.gastro.2023.12.019</pub-id><pub-id pub-id-type="medline">38151192</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Imel</surname><given-names>ZE</given-names> </name><name name-style="western"><surname>Creed</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kious</surname><given-names>B</given-names> </name><name name-style="western"><surname>Althoff</surname><given-names>T</given-names> </name><name name-style="western"><surname>Atzil-Slonim</surname><given-names>D</given-names> </name><name name-style="western"><surname>Srikumar</surname><given-names>V</given-names> </name></person-group><article-title>A framework for automation in psychotherapy</article-title><source>Curr Dir Psychol Sci</source><year>2026</year><month>04</month><volume>35</volume><issue>2</issue><fpage>66</fpage><lpage>76</lpage><pub-id pub-id-type="doi">10.1177/09637214251386047</pub-id><pub-id pub-id-type="medline">41426204</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Carroll</surname><given-names>C</given-names> </name><name name-style="western"><surname>Patterson</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wood</surname><given-names>S</given-names> </name><name name-style="western"><surname>Booth</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rick</surname><given-names>J</given-names> </name><name name-style="western"><surname>Balain</surname><given-names>S</given-names> </name></person-group><article-title>A conceptual framework for implementation fidelity</article-title><source>Implement Sci</source><year>2007</year><month>12</month><volume>2</volume><issue>1</issue><fpage>40</fpage><pub-id pub-id-type="doi">10.1186/1748-5908-2-40</pub-id><pub-id pub-id-type="medline">18053122</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kottner</surname><given-names>J</given-names> </name><name name-style="western"><surname>Audig&#x00E9;</surname><given-names>L</given-names> </name><name name-style="western"><surname>Brorson</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Guidelines for Reporting Reliability and Agreement Studies (GRRAS) were proposed</article-title><source>J Clin Epidemiol</source><year>2011</year><month>01</month><volume>64</volume><issue>1</issue><fpage>96</fpage><lpage>106</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2010.03.002</pub-id><pub-id pub-id-type="medline">21130355</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gallifant</surname><given-names>J</given-names> </name><name name-style="western"><surname>Afshar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ameen</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The TRIPOD-LLM reporting guideline for studies using large language models</article-title><source>Nat Med</source><year>2025</year><month>01</month><volume>31</volume><issue>1</issue><fpage>60</fpage><lpage>69</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03425-5</pub-id><pub-id pub-id-type="medline">39779929</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reyes-Rodr&#x00ED;guez</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Mej&#x00ED;a-Trujillo</surname><given-names>J</given-names> </name><name name-style="western"><surname>P&#x00E9;rez-G&#x00F3;mez</surname><given-names>A</given-names> </name><name name-style="western"><surname>Cardozo</surname><given-names>F</given-names> </name><name name-style="western"><surname>Pinto</surname><given-names>C</given-names> </name></person-group><article-title>Effectiveness of a brief intervention based on motivational interviewing in Colombian adolescents [Article in Portuguese]</article-title><source>Psic: Teor E Pesq</source><year>2018</year><month>01</month><day>8</day><volume>33</volume><fpage>e33421</fpage><pub-id pub-id-type="doi">10.1590/0102.3772e33421</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reyes-Rodr&#x00ED;guez</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Pinto-G&#x00F3;mez</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Cardozo-Mac&#x00ED;as</surname><given-names>F</given-names> </name><name name-style="western"><surname>P&#x00E9;rez-G&#x00F3;mez</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mej&#x00ED;a-Trujillo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Toro-Berm&#x00FA;dez</surname><given-names>J</given-names> </name></person-group><article-title>Evaluation of the prevention program &#x201C;brief intervention based on motivational interviewing&#x201D; in Colombian adolescents</article-title><source>Int J Ment Health Addiction</source><year>2020</year><month>04</month><volume>18</volume><issue>2</issue><fpage>471</fpage><lpage>481</lpage><pub-id pub-id-type="doi">10.1007/s11469-019-0057-3</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="web"><article-title>IBEM&#x2014;Intervenci&#x00F3;n Breve Basada en Entrevista Motivacional [Article in Spanish]</article-title><source>Nuevos Rumbos</source><year>2025</year><access-date>2026-02-03</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://nuevosrumbos.org/website/post?id=581">https://nuevosrumbos.org/website/post?id=581</ext-link></comment></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="web"><article-title>Introducing GPT-4.1 in the API</article-title><source>OpenAI</source><year>2025</year><access-date>2026-08-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openai.com/index/gpt-4-1/">https://openai.com/index/gpt-4-1/</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Colvin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jolibois</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ramezani</surname><given-names>H</given-names> </name></person-group><article-title>Pydantic validation</article-title><source>GitHub</source><year>2026</year><access-date>2026-08-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/pydantic/pydantic">https://github.com/pydantic/pydantic</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="web"><source>Python</source><year>2023</year><access-date>2026-08-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.python.org/">https://www.python.org/</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="web"><person-group person-group-type="author"><collab>The Pandas Development Team</collab></person-group><article-title>Pandas-dev/pandas: pandas</article-title><source>Zenodo</source><year>2025</year><access-date>2026-08-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/16918803">https://zenodo.org/records/16918803</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Richardson</surname><given-names>N</given-names> </name><name name-style="western"><surname>Cook</surname><given-names>I</given-names> </name><name name-style="western"><surname>Crane</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Apache/arrow</article-title><source>GitHub</source><year>2025</year><access-date>2026-08-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/apache/arrow/">https://github.com/apache/arrow/</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="web"><article-title>OpenAI Python library</article-title><source>GitHub</source><access-date>2026-08-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/openai/openai-python">https://github.com/openai/openai-python</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name></person-group><article-title>Weighted kappa: nominal scale agreement with provision for scaled disagreement or partial credit</article-title><source>Psychol Bull</source><year>1968</year><month>10</month><volume>70</volume><issue>4</issue><fpage>213</fpage><lpage>220</lpage><pub-id pub-id-type="doi">10.1037/h0026256</pub-id><pub-id pub-id-type="medline">19673146</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fleiss</surname><given-names>JL</given-names> </name></person-group><article-title>Measuring nominal scale agreement among many raters</article-title><source>Psychol Bull</source><volume>76</volume><issue>5</issue><fpage>378</fpage><lpage>382</lpage><pub-id pub-id-type="doi">10.1037/h0031619</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Landis</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Koch</surname><given-names>GG</given-names> </name></person-group><article-title>The measurement of observer agreement for categorical data</article-title><source>Biometrics</source><year>1977</year><month>03</month><volume>33</volume><issue>1</issue><fpage>159</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.2307/2529310</pub-id><pub-id pub-id-type="medline">843571</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Proctor</surname><given-names>E</given-names> </name><name name-style="western"><surname>Silmere</surname><given-names>H</given-names> </name><name name-style="western"><surname>Raghavan</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Outcomes for implementation research: conceptual distinctions, measurement challenges, and research agenda</article-title><source>Adm Policy Ment Health</source><year>2011</year><month>03</month><volume>38</volume><issue>2</issue><fpage>65</fpage><lpage>76</lpage><pub-id pub-id-type="doi">10.1007/s10488-010-0319-7</pub-id><pub-id pub-id-type="medline">20957426</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moyers</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Manuel</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Hendrickson</surname><given-names>SML</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>WR</given-names> </name></person-group><article-title>Assessing competence in the use of motivational interviewing</article-title><source>J Subst Abuse Treat</source><year>2005</year><month>01</month><volume>28</volume><issue>1</issue><fpage>19</fpage><lpage>26</lpage><pub-id pub-id-type="doi">10.1016/j.jsat.2004.11.001</pub-id><pub-id pub-id-type="medline">15723728</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schoenwald</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Garland</surname><given-names>AF</given-names> </name><name name-style="western"><surname>Chapman</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Frazier</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Sheidow</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Southam-Gerow</surname><given-names>MA</given-names> </name></person-group><article-title>Toward the effective and efficient measurement of implementation fidelity</article-title><source>Adm Policy Ment Health</source><year>2011</year><month>01</month><volume>38</volume><issue>1</issue><fpage>32</fpage><lpage>43</lpage><pub-id pub-id-type="doi">10.1007/s10488-010-0321-0</pub-id><pub-id pub-id-type="medline">20957425</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>Y</given-names> </name></person-group><article-title>LLM-based annotation and token-augmented modeling for emotional tone classification in online cancer peer-support posts</article-title><source>PLOS Digit Health</source><year>2026</year><month>05</month><volume>5</volume><issue>5</issue><fpage>e0001235</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0001235</pub-id><pub-id pub-id-type="medline">42213728</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Checklist 1</label><p>GRRAS checklist.</p><media xlink:href="ai_v5i1e95964_app1.docx" xlink:title="DOCX File, 19 KB"/></supplementary-material><supplementary-material id="app2"><label>Checklist 2</label><p>TRIPOD-LLM checklist.</p><media xlink:href="ai_v5i1e95964_app2.pdf" xlink:title="PDF File, 177 KB"/></supplementary-material></app-group></back></article>