<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e88581</article-id><article-id pub-id-type="doi">10.2196/88581</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Correctness, Harmfulness, and Diversity of Large Language Models for Colonoscopy Preparation Assistance: Comparative Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Kaumenova</surname><given-names>Tomiris</given-names></name><degrees>BA</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chakraborty</surname><given-names>Subhankar</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Fosler-Lussier</surname><given-names>Eric</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gofar</surname><given-names>Kebire</given-names></name><degrees>MD, MPH</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Metcalf</surname><given-names>Isaiah</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Perrault</surname><given-names>Andrew</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>White</surname><given-names>Michael</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Linguistics, The Ohio State University</institution><addr-line>1712 Neil Ave</addr-line><addr-line>Columbus</addr-line><addr-line>OH</addr-line><country>United States</country></aff><aff id="aff2"><institution>Division of Gastroenterology, Hepatology and Nutrition, Department of Internal Medicine, The Ohio State University Wexner Medical Center</institution><addr-line>Columbus</addr-line><addr-line>OH</addr-line><country>United States</country></aff><aff id="aff3"><institution>Department of Computer Science &#x0026; Engineering, The Ohio State University</institution><addr-line>Columbus</addr-line><addr-line>OH</addr-line><country>United States</country></aff><aff id="aff4"><institution>Department of Internal Medicine, Brookdale University Hospital Medical Center</institution><addr-line>Brooklyn</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff5"><institution>College of Medicine, The Ohio State University</institution><addr-line>Columbus</addr-line><addr-line>OH</addr-line><country>United States</country></aff><aff id="aff6"><institution>Novant Health New Hanover Regional Medical Center</institution><addr-line>Wilmington</addr-line><addr-line>NC</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Bhatt</surname><given-names>Manish</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Khasnavis</surname><given-names>Nithisha</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Tomiris Kaumenova, BA, Department of Linguistics, The Ohio State University, 1712 Neil Ave, Columbus, OH, 43210, United States; <email>kaumenova.1@osu.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>4</day><month>8</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e88581</elocation-id><history><date date-type="received"><day>27</day><month>11</month><year>2025</year></date><date date-type="rev-recd"><day>29</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>29</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Tomiris Kaumenova, Subhankar Chakraborty, Eric Fosler-Lussier, Kebire Gofar, Isaiah Metcalf, Andrew Perrault, Michael White. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 4.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e88581"/><abstract><sec><title>Background</title><p>Colorectal cancer is a leading cause of cancer-related deaths in the United States, and colonoscopy remains the gold standard for early detection and prevention. However, many procedures are postponed due to inadequate bowel preparation, a preventable failure often caused by patients&#x2019; difficulty in understanding and following written prep instructions. Prior interventions such as reminder apps and instructional videos have improved adherence only modestly, largely because they cannot answer patient-specific questions. Recent advances in large language models (LLMs) raise the possibility of developing conversational assistants that can provide interactive support to patients in procedure preparation.</p></sec><sec><title>Objective</title><p>This study evaluated the correctness, harmfulness, and diversity of synthetic dialogues generated by leading LLMs acting as both simulated AI Coaches and patients for colonoscopy preparation.</p></sec><sec sec-type="methods"><title>Methods</title><p>Five leading LLMs&#x2014;OpenAI&#x2019;s o3, GPT-4.1, and GPT-5.1; Meta&#x2019;s Llama 3.3 70B; and Mistral&#x2019;s Large-2411&#x2014;were used to generate 250 patient-AI Coach dialogues per model. Dialogues consisted of 3 to 7 question-answer pairs concerning diet, medications, and other prep-related topics. A multiprompt, multiquestion approach was designed to elicit diverse patient questions, and an error taxonomy was established to assess model capabilities in responding to questions. Human raters, including 3 medical experts, evaluated the generated questions for difficulty and the responses for correctness, error type, and potential harmfulness. Automatic evaluation using an LLM-as-a-judge approach complemented human evaluation. Question diversity was assessed using lexical diversity metrics (Distinct-1 and Distinct-2) and entropy. In addition, we evaluated a safety filtering mechanism in which responses judged incorrect by an automated evaluator were replaced with a deferral message instructing patients to contact their health care provider. Differences in response correctness across models were evaluated using permutation tests conducted at the dialogue level. Interrater agreement among human evaluators was assessed using the Gwet AC1 statistic. The study was conducted between May and September 2025.</p></sec><sec sec-type="results"><title>Results</title><p>Automatic evaluation results closely aligned with human judgments: leading models approached but did not achieve adequate performance. Closed-weight models (GPT-5.1, GPT-4.1, and o3) outperformed open-weight models (Llama and Mistral) on correctness, with the reasoning models (GPT-5.1 and o3) performing best. This turn-level ranking was preserved under the supplementary single-prompt baseline, although dialogue-level rankings differed. All models produced harmful errors, primarily due to omissions or misinterpretations of prep instructions. The multiprompt generation strategy substantially increased the diversity of patient questions compared with a single-prompt baseline. Applying an automated safety filter reduced overall error rates but failed to eliminate harmful responses.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Although LLMs demonstrate strong potential to support colonoscopy preparation, none are yet reliable enough for unsupervised deployment in patient-facing contexts. Persistent harmful errors and the limited effectiveness of simple filtering mechanisms highlight the need for improved instruction adherence, stronger safety mechanisms, and validation using real patient queries.</p></sec></abstract><kwd-group><kwd>colorectal cancer</kwd><kwd>colonoscopy</kwd><kwd>health communication</kwd><kwd>virtual assistants</kwd><kwd>large language models</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Colorectal cancer is the fourth leading cause of cancer-related deaths in the United States [<xref ref-type="bibr" rid="ref1">1</xref>], and colonoscopy remains the gold standard for early detection and prevention. Tens of thousands of colonoscopy and endoscopy procedures are performed each year at Ohio State&#x2019;s Wexner Medical Center. However, despite their efficacy, around 20% of these tests are postponed because patients have not read, understood, and correctly carried out prep instructions. Inadequate colonoscopy prep has economic, health-related, and social costs. Economic costs include missed days from taking the prep, taking time off for the procedure, travel costs to get to the appointment and back, and the cost of having an accompanying person to drive (which is a requirement for the procedure). For the health care system, they lead to a missed appointment, incurring resultant wasted resources. Health-related costs include potentially reduced compliance with screening guidelines (only about 70% comply [<xref ref-type="bibr" rid="ref2">2</xref>]) and thus risk of missing polyps. Social costs include frustration on the part of patients, in addition to patients sharing their experience with others, which can deter others from undergoing colonoscopy.</p><p>A key driver of this preventable problem is information overload: patients receive lengthy and complex written instructions days or weeks before their procedure, making it difficult to recall and correctly execute each step at the right time. For example, patients are given the information sheets that typically instruct them to drink half of a prep solution for clearing out the colon at 6 PM on the evening before their scheduled procedure and the other half 6 hours before the procedure time. However, some patients nevertheless show up for their procedure with the second half of their prep solution in hand&#x2014;falsely assuming that they are supposed to take it after arriving at the procedure facility&#x2014;and end up having to reschedule the procedure.</p><p>Past interventions, such as reminder apps [<xref ref-type="bibr" rid="ref3">3</xref>], instructional videos, and automated text messages [<xref ref-type="bibr" rid="ref4">4</xref>], have improved prep adherence only modestly [<xref ref-type="bibr" rid="ref5">5</xref>], largely because they lack the capacity for interactive question answering. The ability to answer questions appears to be critical: when automated systems could not respond to patient questions [<xref ref-type="bibr" rid="ref6">6</xref>], improvements in adherence disappeared [<xref ref-type="bibr" rid="ref7">7</xref>]. Thus, a conversational assistant that can safely respond to patient queries is a promising next step in supporting patients during colonoscopy preparation.</p><p>Recent progress in large language models (LLMs) makes this prospect newly feasible. Frontier models such as o3 and Med-PaLM 2 [<xref ref-type="bibr" rid="ref8">8</xref>] have demonstrated strong reasoning on clinical benchmarks, including OpenAI&#x2019;s HealthBench [<xref ref-type="bibr" rid="ref9">9</xref>] and MedQA [<xref ref-type="bibr" rid="ref10">10</xref>]. If these models can perform complex diagnostic reasoning, a natural question arises: how well can they handle the simpler task of colonoscopy preparation coaching? This task primarily tests prompt adherence, prep instruction understanding, temporal and common sense reasoning, and dialogue communication, rather than clinical inference. Our study attempts to address this question. Building on prior work by Arya et al [<xref ref-type="bibr" rid="ref11">11</xref>], which introduced a neuro-symbolic conversational guide for colonoscopy prep, we explore how LLMs perform on the same challenge. Our preliminary experiments show that recent LLMs substantially outperform earlier models, particularly on temporal reasoning and conversational abilities, which are core difficulties in the task. This motivates a systematic evaluation of LLMs.</p><p>A growing body of work has emerged that explores capabilities and limitations of LLMs in medical question-answering tasks and patient-facing contexts. In nutrition, Sun et al [<xref ref-type="bibr" rid="ref12">12</xref>] have evaluated ChatGPT as an AI dietitian for type 2 diabetes. In mental health, LLMs were assessed on postpartum depression frequently asked questions (FAQs), with responses evaluated using the GRADE (Grading of Recommendations Assessment, Development, and Evaluation) framework [<xref ref-type="bibr" rid="ref13">13</xref>]. In oncology, LLMs were tested on both FAQs [<xref ref-type="bibr" rid="ref14">14</xref>] and patient queries in an electronic patient portal [<xref ref-type="bibr" rid="ref15">15</xref>]. One study found that health care professionals evaluated chatbot responses to patient questions from an online forum as empathetic and high quality [<xref ref-type="bibr" rid="ref16">16</xref>], and other work has shown that patients were more satisfied with LLM-generated responses than with clinicians&#x2019; responses to their questions asked through electronic health records (EHR) [<xref ref-type="bibr" rid="ref17">17</xref>]. Another comprehensive study spanning 17 specialties [<xref ref-type="bibr" rid="ref18">18</xref>] evaluated LLM responses to physician-generated questions for correctness and completeness. Relatedly, the CRAFT-MD (Conversational Reasoning Assessment Framework for Testing in Medicine) framework [<xref ref-type="bibr" rid="ref19">19</xref>] demonstrated how simulated patient-provider interactions can be used to systematically evaluate LLMs across a wide range of clinical tasks. These studies highlight the need for evaluating LLMs in specialized patient-facing and question-answering contexts, such as colonoscopy preparation, and show that such assessments can be done via simulation frameworks, where LLMs act as medical assistants.</p><p>In this study, we selected leading LLMs to generate synthetic dialogues, where models simulated both patients and &#x201C;AI Coaches&#x201D; (ie, colonoscopy prep assistants). These dialogues were evaluated by both human raters and LLM-based raters, enabling us to evaluate not only the dialogue quality but also the ability of LLMs to stand in for human raters. The main benefit of synthetic data is the preservation of patient privacy and safety beyond other advantages [<xref ref-type="bibr" rid="ref20">20</xref>]. Although LLM-generated questions may not fully capture the complexity and variability of questions that real patients may ask, they cover a broad range of topics, offering a challenge and a testing ground for models in this and other tasks [<xref ref-type="bibr" rid="ref21">21</xref>]. The study evaluated dialogues along 2 dimensions: (1) diversity and difficulty of patient questions and (2) correctness of AI Coach responses relative to prep instructions. We aimed to better understand how close current LLMs are to serving as safe and reliable conversational assistants for real-world health care communication tasks in the context of colonoscopy procedure.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Dialogue Generation</title><p>Five leading LLMs with strong reasoning and instruction-following capabilities were selected for dialogue generation: OpenAI&#x2019;s GPT-4.1, o3, and GPT-5.1; Meta&#x2019;s Llama 3.3 70B Instruct; and Mistral&#x2019;s Large-2411. GPT-4.1, o3, and GPT-5.1 are closed-weight models, whereas Mistral Large and Llama 3.3 are open-weight models. Among these, o3 and GPT-5.1 use explicit reasoning. The study was conducted from May to September 2025. GPT-5.1 was released after our data generation and human evaluation were completed. Accordingly, GPT-5.1 dialogues were generated in a separate, later phase of the study. Therefore, we report results for this newer model from automatic evaluation only.</p><p>Dialogue generation was conducted in a turn-by-turn manner, with separate prompts for patient questions and AI Coach responses (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). An example dialogue is provided in <xref ref-type="table" rid="table1">Table 1</xref>. For AI Coach responses, we incorporated few-shot dialogues and chain-of-thought reasoning [<xref ref-type="bibr" rid="ref22">22</xref>], as preliminary experiments indicated that these strategies reduced errors. Few-shot examples modeled fully correct responses and illustrated how to avoid common error types. The AI Coach prompt contained full prep instructions [<xref ref-type="bibr" rid="ref23">23</xref>] converted into raw text (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Patient prompts were designed to encourage diversity and realism, so at each patient&#x2019;s turn, one of several prompt types was selected at random:</p><list list-type="bullet"><list-item><p>Food-related prompts generated questions about eating various foods at a specific time before the procedure or about mistakenly eating restricted foods.</p></list-item><list-item><p>Theme-based prompts generated questions on a specific topic unrelated to foods, such as sedation, allergies, and medication.</p></list-item><list-item><p>Follow-up prompts generated questions based on the previous turn.</p></list-item></list><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Dialogue example: 1. Date: 12 days before procedure. Model: GPT-4.1 Instruct.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Turns</td><td align="left" valign="bottom">Note</td></tr></thead><tbody><tr><td align="left" valign="top">AI Coach: Hello! I&#x2019;m here to help you prepare for your colonoscopy. What questions or concerns can I assist you with today?</td><td align="left" valign="top">Greeting.</td></tr><tr><td align="left" valign="top">Patient: Can I still eat spinach salad with vinaigrette?</td><td align="left" valign="top">Not easily answerable from instructions.</td></tr><tr><td align="left" valign="top">AI Coach: Yes, you can still eat spinach salad with vinaigrette. Dietary changes, such as avoiding leafy greens like spinach, start 5 days before your procedure.</td><td align="left" valign="top">Correct: reminds a patient that spinach is restricted within a 5-day window.</td></tr><tr><td align="left" valign="top">Patient: Got it, thanks for clarifying. What exactly is in the bowel prep drink?</td><td align="left" valign="top">Easily answerable from instructions.</td></tr><tr><td align="left" valign="top">AI Coach: Your bowel prep will include MiraLAX (polyethylene glycol 3350, 8.3 oz or 238 g), 4 Dulcolax tablets (5 mg each), and 64 oz of sports drink (not red or purple). These ingredients are mixed to clean your colon before the procedure.</td><td align="left" valign="top">Correct.</td></tr><tr><td align="left" valign="top">Patient: Will the MiraLAX irritate my ulcerative colitis?</td><td align="left" valign="top">Unanswerable from instructions.</td></tr><tr><td align="left" valign="top">AI Coach: Please contact your health care provider to discuss your specific medical condition and get guidance on your bowel prep.</td><td align="left" valign="top">Correct: directs the patient to a provider.</td></tr></tbody></table></table-wrap><p>Each prompt generated 10 questions per turn, and one of them was selected at random to continue the dialogue. This is a simple approach to addressing the well-known mode collapse problem [<xref ref-type="bibr" rid="ref24">24</xref>], where repeatedly prompting a model produces a limited set of highly similar outputs. For comparison, we also generated dialogues for each model using a baseline prompting approach, where we used a single simple prompt that produced only one question per turn. Each dialogue consisted of 3 to 7 turns (question-answer pairs), with the number of turns determined uniformly at random. A total of 250 dialogues were generated by each model using the 2 prompting approaches, with the same model acting as both the patient and the AI Coach in a dialogue. To assess potential echo-chamber effects, we performed a supplementary crossed-model analysis in which patient questions generated by Llama were answered by model o3.</p><p>In the prep instructions, dietary restrictions begin 5 days before the procedure, marking the start of the preparation period most relevant to patients. Accordingly, dialogues were assigned a time point between 1 and 21 days before the procedure, with dialogues set within the 5-day window occurring twice as frequently as those set earlier, as these tend to involve more challenging or clinically relevant questions, and patients tend to ask questions closer to the procedure. Model correctness was later stratified by short-term (1&#x2010;5 days) and longer-term (6&#x2010;21 days) windows.</p></sec><sec id="s2-2"><title>Human Evaluation</title><p>Four human raters participated in the dialogue evaluation: 1 attending physician (expert), 1 medical resident, 1 medical student (experienced raters), and 1 layperson (the first author). For each model (except GPT-5.1, as noted above), 50 dialogues were sampled at random for evaluation. A subset of 50 dialogues was annotated by all 4 raters to determine interrater agreement, with dialogues evenly distributed across models. The rest of the dialogues were split evenly between experienced raters. All analyses reported are based on expert and experienced rater annotations. Additionally, a subset of dialogues was cross-annotated by the lay rater to enable comparison between lay and experienced evaluations. To support consistency, raters were provided with detailed evaluation guidelines, which were refined following a pilot trial. All raters except the lead author were blinded to the model source of the dialogues.</p><p>Raters were instructed to determine whether each AI Coach response was correct or incorrect, with reference to prep instructions. Incorrect responses were classified into one of the following categories (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>):</p><list list-type="bullet"><list-item><p>Temporal errors: inconsistencies with the procedure timeline</p></list-item><list-item><p>Extraneous information: inclusion of content outside the scope of prep instructions (subcategorized as factually correct vs factually incorrect)</p></list-item><list-item><p>Reasoning errors: contradictions with prep instructions or faulty reasoning (eg, about restricted foods)</p></list-item><list-item><p>Omissions: exclusion of essential or helpful information</p></list-item><list-item><p>Other errors: irrelevant, disfluent, or ambiguous responses</p></list-item></list><p>Each error was also judged for harmfulness. An error was classified as harmful if following the advice could plausibly result in inadequate bowel preparation, violation of medication or fasting restrictions, or a canceled procedure. Errors were classified as harmless if they introduced inaccuracies that would not affect preparation quality, procedure scheduling, or patient safety (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). Disagreements among raters were resolved by majority vote; in cases without a consensus, the rating of the most experienced rater was used. Responses may contain multiple errors and were considered correct only if they contained none.</p><p>The correctness metric, which emphasizes strict adherence to prep instructions, was supplemented with absolute correctness, in which responses containing factually correct but extraneous information beyond the preparation instructions were merged with correct responses rather than treated as errors. Responses containing factually correct but extraneous information are harmless by definition.</p><p>Unlike AI Coach responses, patient questions were not judged for correctness because, in practice, patients may ask irrelevant or ambiguous questions, and providers are still expected to respond appropriately. Instead, 30 questions from each model were sampled at random and categorized into 1 of 3 levels: (1) easily answerable from prep instructions; (2) not easily answerable, such as those that require reasoning about the timeline or fiber content of various foods; (3) not answerable at all (eg, sedation details or anxiety management), which require deferral to a provider. This classification aimed to determine whether models were presented with questions of varying degrees of difficulty, as would be expected in a real-life scenario.</p></sec><sec id="s2-3"><title>Automatic Evaluation</title><p>We used an LLM-as-a-judge approach to evaluate a large set of dialogues, in which each AI Coach response was independently evaluated by a language model. The evaluation prompt (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) included the full prep instructions, few-shot examples of common errors and correct responses, and it required the model to provide an explanation whenever it judged a response as incorrect. These explanations were retained for qualitative error analysis.</p><p>Four evaluator models were tested: DeepSeek-R1, OpenAI&#x2019;s GPT-5.1 and o3, and Meta&#x2019;s Llama 3.3 70B. These models were selected to represent both closed-weight and open-weight models with strong reasoning performance. To determine the most reliable automated evaluator, we compared evaluator predictions with human judgments. DeepSeek-R1 demonstrated the best performance and was therefore selected as the primary automated evaluator for reporting large-scale results (<xref ref-type="table" rid="table2">Table 2</xref>). Although prior work suggests that LLM-based evaluators may exhibit self-bias [<xref ref-type="bibr" rid="ref25">25</xref>], none of the evaluated dialogues in our study were generated by DeepSeek-R1, mitigating that risk.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Performance of large language models as automated error predictors in synthetic colonoscopy preparation dialogues between simulated patients and AI Coaches (N=252)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Accuracy, n (%)</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">Llama 3.3 70B</td><td align="left" valign="top">199 (79.0)</td><td align="left" valign="top">0.63</td><td align="left" valign="top">0.40</td><td align="left" valign="top">0.49</td></tr><tr><td align="left" valign="top">GPT-5.1</td><td align="left" valign="top">207 (82.1)</td><td align="left" valign="top">0.45</td><td align="left" valign="top">0.55</td><td align="left" valign="top">0.49</td></tr><tr><td align="left" valign="top">o3</td><td align="left" valign="top">213 (84.5)</td><td align="left" valign="top">0.51</td><td align="left" valign="top">0.58</td><td align="left" valign="top">0.54</td></tr><tr><td align="left" valign="top">DeepSeek-R1</td><td align="left" valign="top">217 (86.1)</td><td align="left" valign="top">0.56</td><td align="left" valign="top">0.58</td><td align="left" valign="top">0.57</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Fifty randomly selected dialogues were cross-evaluated by expert human raters, whose annotations served as the gold standard. Model predictions of response correctness were compared against these expert ratings at the turn level. Cochran <italic>Q</italic> test detected no differences among evaluators.</p></fn></table-wrap-foot></table-wrap><p>We also implemented a simple safety filtering mechanism based on the best-performing automated evaluator. If the evaluator classified a response as incorrect, the original response was replaced with a deferral message instructing the patient to contact their health care provider for clarification. If the response was classified as correct, the original response was retained. Error and harmfulness rates were then recomputed using the modified set of responses to estimate the potential impact of this filtering mechanism. This procedure was performed offline as a postprocessing step.</p><p>Additionally, we assessed the diversity of patient questions across models using the type-token ratio (the percentage of unique unigrams and bigrams, referred to as Distinct-1 and Distinct-2, respectively) and the entropy measure. We compared question diversity generated using our multiprompt, 10-question strategy with a single-prompt, single-question approach, which served as the baseline. The most frequent tokens are listed in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p></sec><sec id="s2-4"><title>Statistical Analysis</title><p>We estimated the approximate sample size of 1228 responses per model for the automatic evaluation using a chi-square power analysis with a significance level of 0.05 and statistical power of 0.80. Pilot estimates of the observed proportions yielded a Cohen effect size of 0.094. Assuming an average of 5 responses per dialogue, 1228 responses corresponded to 245.6 dialogues per model, which was rounded to 250. A Bonferroni-adjusted significance level was applied to maintain the desired statistical power when conducting multiple comparisons. Two models (Llama 3.3 70B and GPT-5.1) produced slightly fewer responses than the planned target (1220 and 1219, respectively), representing a deviation of less than 1% from the planned sample size, which does not significantly affect the intended statistical power.</p><p>Chi-square tests and tests of proportions treat dialogue turns as independent observations. Therefore, to confirm the statistical significance of differences in turn-level accuracy between 2 models (Model A and Model B) without assuming that turns in the same dialogue were independent, we used a nonparametric permutation test at the dialogue level with a custom turn-level accuracy statistic. For each dialogue, the number of correct turns and total turns was calculated. The observed test statistic, <italic>T</italic><sub>obs</sub>, was calculated as the difference in turn-level accuracy between Model A and Model B, which was determined by summing and dividing the correct and total turns across all dialogues for each model. We then generated an empirical null distribution by repeatedly shuffling the dialogue-level group labels (n=50,000) and recalculating the statistic for each permuted set. The 2-sided <italic>P</italic> value was determined as the proportion of permutations in which the absolute value of the permuted statistic was greater than or equal to |<italic>T</italic><sub>obs</sub>|. Such pairwise permutation tests were conducted against the best-performing model with a Holm-Bonferroni&#x2013;adjusted significance level of .05.</p><p>For human evaluation, we estimated the sample size required to detect significant interrater agreement using the Bloch and Kraemer formula for Krippendorff &#x03B1; [<xref ref-type="bibr" rid="ref26">26</xref>], assuming a significance level of .05. However, we did not anticipate such high correctness rates observed in our data and instead calculated Gwet AC1 as the measure of interrater agreement. This statistic is more robust to skewed category distributions [<xref ref-type="bibr" rid="ref27">27</xref>] and is reported in the <italic>Discussion</italic> section.</p></sec><sec id="s2-5"><title>Ethical Considerations</title><p>This study did not involve human participants; therefore, institutional board review approval was not required. All data analyzed in this study were fully synthetic and did not contain any real patient information.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Question Difficulty and Diversity</title><p>Human evaluation of question difficulty (<xref ref-type="table" rid="table3">Table 3</xref>) showed that questions were well balanced across difficulty levels for all models, with o3 and Mistral Large generating slightly more challenging or unanswerable questions. Automatic diversity metrics, such as entropy and Distinct-1 or Distinct-2 (<xref ref-type="table" rid="table4">Table 4</xref>), indicated that the multiprompt, multiquestion generation strategy produced more diverse patient questions than the single-prompt, single-question baseline for all models. Both o3 and GPT-5.1 produced the most lexically diverse patient questions, achieving the highest Distinct-1 or Distinct-2 and entropy scores among all models. <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref> lists the most frequent tokens across models, revealing a clear qualitative pattern: stronger models such as o3, GPT-5.1, and GPT-4.1 generated more specific terms (eg, chicken, coffee, white), whereas Llama and its baseline relied more on generic ones (eg, colonoscopy, procedure). Although Mistral produced a range of specific items (eg, popcorn, wine, salad), its token frequencies were more uneven, whereas the stronger models exhibited more balanced lexical distributions.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Human evaluation of patient question difficulty in colonoscopy preparation dialogues generated by different large language models (N=30 per model)<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Easily answerable, n</td><td align="left" valign="bottom">Not easily answerable, n</td><td align="left" valign="bottom">Unanswerable, n</td></tr></thead><tbody><tr><td align="left" valign="top">Mistral Large</td><td align="left" valign="top">6</td><td align="left" valign="top">12</td><td align="left" valign="top">12</td></tr><tr><td align="left" valign="top">Llama 3.3 70B</td><td align="left" valign="top">10</td><td align="left" valign="top">8</td><td align="left" valign="top">12</td></tr><tr><td align="left" valign="top">GPT-4.1</td><td align="left" valign="top">11</td><td align="left" valign="top">10</td><td align="left" valign="top">9</td></tr><tr><td align="left" valign="top">o3</td><td align="left" valign="top">5</td><td align="left" valign="top">12</td><td align="left" valign="top">13</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Patient questions were generated using a multiprompt, 10 questions-per-prompt strategy. Questions were categorized by human raters according to whether they were directly answerable from the colonoscopy preparation instructions, required indirect reasoning, or were unanswerable based on the provided instructions.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Automatic evaluation of question diversity in colonoscopy preparation dialogues generated by different large language models<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Distinct-1</td><td align="left" valign="bottom">Distinct-2</td><td align="left" valign="bottom">Entropy</td><td align="left" valign="bottom">N</td></tr></thead><tbody><tr><td align="left" valign="top">Baseline Llama</td><td align="left" valign="top">0.03</td><td align="left" valign="top">0.10</td><td align="left" valign="top">7.22</td><td align="left" valign="top">1227</td></tr><tr><td align="left" valign="top">Llama 3.3 70B</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.25</td><td align="left" valign="top">8.00</td><td align="left" valign="top">1220</td></tr><tr><td align="left" valign="top">Baseline GPT-4.1</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.24</td><td align="left" valign="top">7.90</td><td align="left" valign="top">1264</td></tr><tr><td align="left" valign="top">GPT-4.1</td><td align="left" valign="top">0.11</td><td align="left" valign="top">0.39</td><td align="left" valign="top">8.59</td><td align="left" valign="top">1292</td></tr><tr><td align="left" valign="top">Baseline Mistral</td><td align="left" valign="top">0.09</td><td align="left" valign="top">0.28</td><td align="left" valign="top">7.91</td><td align="left" valign="top">1233</td></tr><tr><td align="left" valign="top">Mistral Large</td><td align="left" valign="top">0.11</td><td align="left" valign="top">0.40</td><td align="left" valign="top">8.70</td><td align="left" valign="top">1250</td></tr><tr><td align="left" valign="top">Baseline GPT-5.1</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.25</td><td align="left" valign="top">8.10</td><td align="left" valign="top">1282</td></tr><tr><td align="left" valign="top">GPT-5.1</td><td align="left" valign="top">0.16</td><td align="left" valign="top">0.47</td><td align="left" valign="top">8.93</td><td align="left" valign="top">1219</td></tr><tr><td align="left" valign="top">Baseline o3</td><td align="left" valign="top">0.12</td><td align="left" valign="top">0.40</td><td align="left" valign="top">9.02</td><td align="left" valign="top">1277</td></tr><tr><td align="left" valign="top">o3</td><td align="left" valign="top">0.16</td><td align="left" valign="top">0.51</td><td align="left" valign="top">9.14</td><td align="left" valign="top">1262</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Diversity metrics, including the proportion of unique unigrams and bigrams (Distinct-1 and Distinct-2, respectively) and lexical entropy, were computed across patient questions. Results include both a single-prompt, single-question baseline and a multiprompt, 10-questions-per-prompt generation strategy for each model.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Overall Response Accuracy</title><p><xref ref-type="table" rid="table5">Table 5</xref> presents the human evaluation results for dialogue-level and turn-level accuracy relative to the colonoscopy preparation instructions. At the dialogue level, o3 achieved the highest accuracy (38/50, 76%), followed by GPT-4.1 (31/50, 62%), Llama 3.3 70B (19/50, 38%), and Mistral Large (9/50, 18%). A similar pattern was observed at the turn level, with o3 (216/231, 93.5%) significantly outperforming Llama 3.3 70B (185/237, 78.1%; <italic>P</italic>&#x003C;.001) and Mistral Large (171/251, 68.1%; <italic>P</italic>&#x003C;.001). The comparison with GPT-4.1 (227/250, 90.8%; <italic>P</italic>=.28) was not significant. When counting responses that were extraneous but still correct, absolute correctness increased only slightly for all models.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Human evaluation of model performance in colonoscopy preparation dialogues generated by different large language models<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Dialogue-level accuracy, n/N (%)</td><td align="left" valign="bottom">Turn-level accuracy, n/N (%)</td><td align="left" valign="bottom">Absolute correctness, n/N (%)</td><td align="left" valign="bottom"><italic>T</italic><sub>obs</sub></td><td align="left" valign="bottom">Adjusted <italic>P</italic> value (vs best)</td></tr></thead><tbody><tr><td align="left" valign="top">Mistral Large</td><td align="left" valign="top">9/50 (18)</td><td align="left" valign="top">171/251 (68.1)</td><td align="left" valign="top">178/251 (70.9)</td><td align="left" valign="top">25.4</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Llama 3.3 70B</td><td align="left" valign="top">19/50 (38)</td><td align="left" valign="top">185/237 (78.1)</td><td align="left" valign="top">189/237 (79.7)</td><td align="left" valign="top">15.4</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">GPT-4.1</td><td align="left" valign="top">31/50 (62)</td><td align="left" valign="top">227/250 (90.8)</td><td align="left" valign="top">231/250 (92.4)</td><td align="left" valign="top">2.7</td><td align="left" valign="top">.28</td></tr><tr><td align="left" valign="top">o3</td><td align="left" valign="top">38/50 (76)</td><td align="left" valign="top">216/231 (93.5)</td><td align="left" valign="top">218/231 (94.4)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Dialogue-level and turn-level accuracy were assessed relative to publicly available colonoscopy preparation instructions from the Wexner Medical Center. Absolute correctness includes responses containing accurate but extraneous information beyond the preparation instructions. Pairwise comparisons of turn-level accuracy against the highest-performing model (o3) were conducted using dialogue-level permutation tests with Holm-Bonferroni correction.</p></fn><fn id="table5fn2"><p><sup>b</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><p>To extend these findings to a larger dataset, we conducted an automatic evaluation using the LLM-as-a-judge approach. We used 4 LLMs as evaluators: DeepSeek-R1, OpenAI&#x2019;s GPT-5.1 and o3, and Meta&#x2019;s Llama 3.3 70B (<xref ref-type="table" rid="table2">Table 2</xref>). We ultimately selected DeepSeek-R1 as the evaluator, as it achieved the highest accuracy and <italic>F</italic><sub>1</sub>-score against the human judgments. The automatic evaluation (<xref ref-type="table" rid="table6">Table 6</xref>) followed similar trends to those observed in the human evaluation. At both the dialogue level (182/250, 72.8%) and turn level (1135/1219, 93.1%), GPT-5.1 was the best-performing model, significantly exceeding o3 (158/250, 63.2% and 1145/1262, 90.7%, respectively; <italic>P</italic>=.04), GPT-4.1 (147/250, 58.8% and 1148/1292, 88.9%, respectively; <italic>P</italic>&#x003C;.001), Llama 3.3 70B (104/250, 41.6% and 990/1220, 81.1%, respectively; <italic>P</italic>&#x003C;.001), and Mistral Large (150/250, 60% and 773/1250, 61.8%, respectively; <italic>P</italic>&#x003C;.001) at the turn level. The supplementary baseline correctness analysis (<xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>) confirmed that GPT-5.1 was the best-performing model under both the single-prompt, single-question baseline and the multiprompt, multiquestion prompting approaches. In a supplementary crossed-model analysis (<xref ref-type="table" rid="table7">Table 7</xref>), o3 achieved slightly higher accuracy when answering Llama-generated patient questions (1127/1220, 92.4%) than when answering its own questions. At the dialogue level, the accuracy increased substantially by approximately 30%, suggesting that incorrect turns were concentrated within a small number of dialogues. In a supplementary stratified analysis (<xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>), all models showed higher accuracy in the 6 to 21-day window at both the turn and dialogue levels. The 1 to 5-day window, sampled more frequently to reflect patient behavior in the real world, was more challenging, likely due to increased clinical complexity during the restriction period.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Automatic evaluation of dialogue-level and turn-level accuracy in colonoscopy preparation dialogues generated by different large language models (LLMs)<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup>.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Dialogue-level accuracy, n/N (%)</td><td align="left" valign="bottom">Turn-level accuracy, n/N (%)</td><td align="left" valign="bottom"><italic>T</italic><sub>obs</sub></td><td align="left" valign="bottom">Adjusted <italic>P</italic> value (vs best)</td></tr></thead><tbody><tr><td align="left" valign="top">Mistral Large</td><td align="left" valign="top">150/250 (60)</td><td align="left" valign="top">773/1250 (61.8)</td><td align="left" valign="top">31.3</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Llama 3.3 70B</td><td align="left" valign="top">104/250 (41.6)</td><td align="left" valign="top">990/1220 (81.1)</td><td align="left" valign="top">12.0</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">GPT-4.1</td><td align="left" valign="top">147/250 (58.8)</td><td align="left" valign="top">1148/1292 (88.9)</td><td align="left" valign="top">4.2</td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top">o3</td><td align="left" valign="top">158/250 (63.2)</td><td align="left" valign="top">1145/1262 (90.7)</td><td align="left" valign="top">2.4</td><td align="left" valign="top">.04</td></tr><tr><td align="left" valign="top">GPT-5.1</td><td align="left" valign="top">182/250 (72.8)</td><td align="left" valign="top">1135/1219 (93.1)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup></td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>Accuracy was determined using an LLM-as-judge approach, with DeepSeek-R1 serving as the automated evaluator. Pairwise comparisons of turn-level accuracy against the highest-performing model (GPT-5.1) were conducted using dialogue-level permutation tests with Holm-Bonferroni correction.</p></fn><fn id="table6fn2"><p><sup>b</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t7" position="float"><label>Table 7.</label><caption><p>Supplementary automatic evaluation of dialogue-level and turn-level accuracy in colonoscopy preparation dialogues for Llama 3.3 70B and o3 under original and crossed dialogue-generation conditions<sup><xref ref-type="table-fn" rid="table7fn1">a</xref></sup>.</p></caption><table id="table7" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Dialogue-level accuracy, n/N (%)</td><td align="left" valign="bottom">Turn-level accuracy, n/N (%)</td></tr></thead><tbody><tr><td align="left" valign="top">o3</td><td align="left" valign="top">158/250 (63.2)</td><td align="left" valign="top">1145/1262 (90.7)</td></tr><tr><td align="left" valign="top">Llama+o3</td><td align="left" valign="top">232/250 (92.8)</td><td align="left" valign="top">1127/1220 (92.4)</td></tr></tbody></table><table-wrap-foot><fn id="table7fn1"><p><sup>a</sup>Accuracy was determined using an LLM-as-judge approach, with DeepSeek-R1 serving as the automated evaluator. Dialogues generated entirely by o3 were compared with a crossed-model condition in which patient questions generated by Llama 3.3 70B were answered by o3.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Error Analysis</title><p>Error breakdowns are reported in <xref ref-type="table" rid="table8">Table 8</xref>. The most frequent error type across models was omission, ranging from 1.7% (4/231) for o3 to 11.6% (29/251) for Mistral Large. Models o3 and GPT-4.1 consistently had the lowest error rates across all categories. The error rates of model o3 remained uniformly below 2% across all error types. <xref ref-type="table" rid="table9">Table 9</xref> shows that both the number and proportion of harmful errors varied notably by model. Mistral Large produced the highest number of total errors but surprisingly a lower proportion of harmful errors (37/114, 32.5%). In contrast, model o3 had the fewest total errors but the highest proportion that were harmful (11/17, 64.7%).</p><table-wrap id="t8" position="float"><label>Table 8.</label><caption><p>Distribution of error types across models in colonoscopy preparation dialogues generated by different large language models based on human evaluation<sup><xref ref-type="table-fn" rid="table8fn1">a</xref></sup>.</p></caption><table id="table8" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Temporal, n/N (%)</td><td align="left" valign="bottom">Extraneous correct, n/N (%)</td><td align="left" valign="bottom">Extraneous incorrect, n/N (%)</td><td align="left" valign="bottom">Reasoning, n/N (%)</td><td align="left" valign="bottom">Omission, n/N (%)</td><td align="left" valign="bottom">Other, n/N (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Mistral Large</td><td align="left" valign="top">26/251 (10.4)</td><td align="left" valign="top">8/251 (3.2)</td><td align="left" valign="top">7/251 (2.8)</td><td align="left" valign="top">27/251 (10.8)</td><td align="left" valign="top">29/251 (11.6)</td><td align="left" valign="top">3/251 (1.2)</td></tr><tr><td align="left" valign="top">Llama 3.3 70B</td><td align="left" valign="top">10/237 (4.2)</td><td align="left" valign="top">4/237 (1.7)</td><td align="left" valign="top">9/237 (3.7)</td><td align="left" valign="top">7/237 (3.0)</td><td align="left" valign="top">23/237 (9.7)</td><td align="left" valign="top">3/237 (1.3)</td></tr><tr><td align="left" valign="top">GPT-4.1</td><td align="left" valign="top">7/250 (2.8)</td><td align="left" valign="top">4/250 (1.6)</td><td align="left" valign="top">3/250 (1.2)</td><td align="left" valign="top">1/250 (0.4)</td><td align="left" valign="top">12/250 (4.8)</td><td align="left" valign="top">1/250 (0.4)</td></tr><tr><td align="left" valign="top">o3</td><td align="left" valign="top">1/231 (0.4)</td><td align="left" valign="top">3/231 (1.3)</td><td align="left" valign="top">4/231 (1.7)</td><td align="left" valign="top">4/231 (1.7)</td><td align="left" valign="top">4/231 (1.7)</td><td align="left" valign="top">1/231 (0.4)</td></tr></tbody></table><table-wrap-foot><fn id="table8fn1"><p><sup>a</sup>Error categories were defined according to predefined annotation guidelines. Fifty dialogues per model were evaluated. Rates are calculated at the turn level, and n denotes the number of turns containing at least 1 error of the specified type.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t9" position="float"><label>Table 9.</label><caption><p>Proportion of harmless vs harmful errors across models in colonoscopy preparation dialogues generated by different large language models, based on human evaluation<sup><xref ref-type="table-fn" rid="table9fn1">a</xref></sup>.</p></caption><table id="table9" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Harmless, n/N (%)</td><td align="left" valign="bottom">Harmful, n/N (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Mistral Large</td><td align="left" valign="top">77/114 (67.5)</td><td align="left" valign="top">37/114 (32.5)</td></tr><tr><td align="left" valign="top">Llama 3.3 70B</td><td align="left" valign="top">28/58 (48.3)</td><td align="left" valign="top">30/58 (51.7)</td></tr><tr><td align="left" valign="top">GPT-4.1</td><td align="left" valign="top">17/28 (60.7)</td><td align="left" valign="top">11/28 (39.3)</td></tr><tr><td align="left" valign="top">o3</td><td align="left" valign="top">6/17 (35.3)</td><td align="left" valign="top">11/17 (64.7)</td></tr></tbody></table><table-wrap-foot><fn id="table9fn1"><p><sup>a</sup>Harmfulness was defined as an error with the potential to cause clinically significant consequences, including procedure cancellation or adverse health outcomes.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Diversity of Patient Questions</title><p>Prompt design had a clear effect on the diversity of patient queries. A simple baseline prompt produced repetitive questions, whereas our multiprompt strategy, where each turn generated 10 candidate questions from food-related, thematic, or follow-up prompts, led to greater lexical and topical variety, helping mitigate the mode collapse problem. Automatic metrics (Distinct-<italic>n</italic> and entropy) confirmed this increase in diversity. Human evaluation further showed that our approach yielded a balanced distribution of questions: some were easily answerable from the preparation instructions, others required indirect reasoning (eg, about timing or food composition), and some were not answerable at all (eg, anxiety or sedation concerns). This suggests that our generation setup exposed models to a realistic range of question types and difficulty, mirroring those encountered in actual patient-provider interactions. Consequently, the resulting dialogues provide a more robust basis for evaluating models&#x2019; performance in this task.</p></sec><sec id="s4-2"><title>Correctness of AI Coach Responses</title><p>This task is relatively straightforward: responses are evaluated against a fixed set of colonoscopy preparation instructions. However, the best-performing models (GPT-4.1, o3, and GPT-5.1) approach but do not reach 100% accuracy. Smaller open-source models (Llama and Mistral) are not a viable alternative due to a substantially higher number of errors. While a fully crossed design could be considered in future work, the preliminary analysis of the Llama and o3 pairing indicates that using the same-model pairing did not artificially boost performance.</p><p>Error patterns revealed important distinctions across models. Temporal errors were much less frequent in the stronger models; however, instruction-following errors (omission, extraneous information, and faulty reasoning) remained prevalent even when temporal accuracy was high, suggesting that models could benefit from further fine-tuning for instruction adherence. Additionally, o3 produced a large proportion of harmful errors despite high overall accuracy, showing that accuracy and safety are not equivalent. In fact, the existence of harmful errors demonstrates the potential risks of deploying even highly accurate models in patient-facing contexts without additional safeguards. This concern may be especially relevant in scenarios where patients already have a solid understanding of preparation instructions, since models could inadvertently misdirect otherwise well-prepared patients. For less-prepared patients, however, model assistance could still be mildly beneficial, although not enough on its own to ensure a successful prep. Patients may need reminders [<xref ref-type="bibr" rid="ref3">3</xref>] to complete preparation steps, which would in turn require a more complex interactive system capable of supporting these mechanisms.</p><p>Common harmful errors involved omissions of critical safety or procedural information. Examples include failing to mention the need for a designated driver at check-in, omitting the 2-hour restriction on liquids (with the exception of small sips of water for medication), or failing to warn about future dietary restrictions (eg, avoiding red or purple liquids and dairy products on the day before the procedure). Other errors concerned medication guidance, such as not advising patients to consult their provider about medication changes or neglecting to state specific rules for oral diabetes medications. In several cases, models failed to instruct patients to contact their provider after mistakenly consuming a prohibited food item. There were also some patient questions phrased in terms of weekdays (eg, &#x201C;this Sunday&#x201D;) rather than in terms of the number of days left before the procedure (eg, &#x201C;3 days before&#x201D;). Such questions create ambiguity that a robust system should detect, but even stronger models were not able to do so.</p><p>Model behavior also differed in style. Mistral&#x2019;s responses were often extremely brief, offering no explanation derived from the instructions and sometimes consisting of single-word replies. Mistral also refused to answer many dietary questions when dialogues were set many days before the procedure and conflated patient and provider roles more often than others. Llama tended to be verbose and occasionally unnatural in phrasing, whereas GPT and o3 produced more natural responses, which were lengthy only when warranted.</p></sec><sec id="s4-3"><title>Evaluation</title><p>Both automatic and human evaluation were essential and complementary. Human evaluation provided insights into error types and their potential harmfulness, while automatic evaluation provided a scalable way to evaluate large numbers of dialogues beyond what would be feasible with expert raters. Interrater agreement for correctness judgments among the human raters was substantial (expert and experienced raters: AC1=0.74, 95% CI 0.68-0.80, percent agreement=0.82; all raters: AC1=0.68, 95% CI 0.62-0.75, percent agreement=0.79), consistent with our expectation that evaluating factual correctness in this task is easier than clinical diagnosis.</p><p>Interestingly, we found systematic differences between DeepSeek-R1 (R1) and the expert raters. R1&#x2019;s strict adherence to prep instructions led it to identify certain erroneous responses that the experts judged acceptable, suggesting that automatic evaluators may in fact be more reliable at guideline and prep fidelity. Some of these cases involved minor issues, such as responses that were incomplete or lacked explanation. In one case, the response model misclassified Tylenol as a nonsteroidal anti-inflammatory drug (NSAID), which constitutes a factual error but not a harmful one, as experts agreed the real safety concern lies with anticoagulants such as Warfarin. In another case, the response model failed to specify that instant Ramen should be made from low-fiber noodles. Although technically correct, this omission was considered inconsequential because most instant Ramen products already meet that criterion. R1 was able to successfully identify these erroneous cases, but it also penalized a few factually correct responses, reflecting an overly rigid standard of adherence.</p><p>These findings do not undermine the earlier results from human evaluation. Importantly, evaluating a large set of responses offline in a spreadsheet differs from interacting with real patients, where pragmatic communication takes precedence. The lay rater showed similar tendencies to R1, prioritizing strict adherence while making only a single factual error (incorrectly believing that gummy bears were allowed on the day before the procedure). This further suggests that factual evaluation of dialogues in reference to prep instructions is a relatively easy task for humans to learn, yet even the strongest models are not perfect at it.</p><p>Regarding false negatives, R1 missed a few temporal and reasoning errors, but no clear pattern was observed among them. In a safety filtering context, such false negatives are more concerning than false positives because they allow incorrect responses to remain unflagged. However, these cases were relatively infrequent and do not change the comparative findings across models.</p></sec><sec id="s4-4"><title>Safety Filter</title><p>Applying an LLM-as-a-judge filter (<xref ref-type="table" rid="table10">Table 10</xref>) to the human-annotated subset of dialogues reduced both the overall error rate and the harmfulness rate for all models. The filter consisted of replacing all AI Coach responses that the LLM judge classified as incorrect with a deferral to a provider, then recomputing turn-level error and harmfulness rate on the responses. The relative improvement was greatest for Mistral Large (error rate reduced from 80/251, 31.9% to 27/251, 10.8%) and Llama 3.3 70B (52/237, 21.9% to 29/237, 12.2%). Model o3 remained the most accurate before and after filtering (15/231, 6.5% to 12/231, 5.2%). Similarly for harmfulness, the filter helped weaker models like Llama and Mistral, but offered little or no benefit for stronger models such as GPT-4.1 or o3, where it removed only one harmful error per model. These improvements should be interpreted with caution, since we do not penalize false provider deferrals in our framework. If such a filter were deployed in patient interactions, it would likely be counterproductive: if patients are redirected to contact a provider for questions that could have been answered from the prep instructions, the usefulness of the AI Coach as an assistant for colonoscopy prep is undermined. Most importantly, the filter failed to eliminate a sufficient number of harmful errors, making it impractical as a safety mechanism.</p><table-wrap id="t10" position="float"><label>Table 10.</label><caption><p>Error and harmfulness rates in the human-annotated subset of colonoscopy preparation dialogues before and after applying a large language model (LLM)-as-a-judge filter<sup><xref ref-type="table-fn" rid="table10fn1">a</xref></sup>.</p></caption><table id="table10" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Filtering</td><td align="left" valign="bottom" colspan="2">Before</td><td align="left" valign="bottom" colspan="2">After</td></tr><tr><td align="left" valign="top">Model</td><td align="left" valign="top">Error rate, n/N (%)</td><td align="left" valign="top">Harmfulness, n/N (%)</td><td align="left" valign="top">Error rate, n/N (%)</td><td align="left" valign="top">Harmfulness, n/N (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Mistral Large</td><td align="left" valign="top">80/251 (31.9)</td><td align="left" valign="top">33/251 (13.1)</td><td align="left" valign="top">27/251 (10.8)</td><td align="left" valign="top">11/251 (4.4)</td></tr><tr><td align="left" valign="top">Llama 3.3 70B</td><td align="left" valign="top">52/237 (21.9)</td><td align="left" valign="top">30/237 (12.7)</td><td align="left" valign="top">29/237 (12.2)</td><td align="left" valign="top">18/237 (7.6)</td></tr><tr><td align="left" valign="top">GPT-4.1</td><td align="left" valign="top">23/250 (9.2)</td><td align="left" valign="top">9/250 (3.6)</td><td align="left" valign="top">15/250 (6)</td><td align="left" valign="top">8/250 (3.2)</td></tr><tr><td align="left" valign="top">o3</td><td align="left" valign="top">15/231 (6.5)</td><td align="left" valign="top">10/231 (4.3)</td><td align="left" valign="top">12/231 (5.2)</td><td align="left" valign="top">9//231 (3.9)</td></tr></tbody></table><table-wrap-foot><fn id="table10fn1"><p><sup>a</sup>The filter replaced AI Coach responses classified as incorrect by the LLM judge with a deferral advising the patient to contact their provider. Rates are calculated at the turn level; N denotes the number of turns containing at least one error and at least one harmful error, respectively.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s4-5"><title>Limitations</title><p>This study has several limitations. First and foremost, the use of synthetic dialogues may limit external validity. The distribution of synthetic patient questions may not fully capture the range or underlying motivations behind real-world patient inquiries. The factors driving patient unpreparedness are not fully understood and can stem from causes beyond informational gaps, such as anxiety or noncompliance, that are difficult to model through prompt-based generation. Additionally, real patient questions can prove to be more challenging, as they can be emotionally loaded, unclear, or fragmented and may require follow-up clarification, although expert raters qualitatively found the synthetic questions to be otherwise plausible. Furthermore, the prompts explicitly encouraged creativity, which may have led to overrepresentation of atypical or edge-case questions that do not fully reflect the distribution of real patient inquiries. Consequently, models might perform differently when confronted with real patient queries along with potential medical comorbidities, and the correctness results reported here may not directly translate to real clinical deployment settings. Future evaluation on real patient-provider communication data will be necessary to determine the true clinical generalizability of our findings.</p><p>Second, although model responses were evaluated for factual accuracy relative to the preparation instructions, other aspects of communication quality, such as empathy, were not assessed. Additionally, while a crossed-model analysis was performed to partially assess potential echo-chamber effects, this was limited to a subset of model pairings and does not exclude such effects for weaker models. In the future, all models should be evaluated against a standardized, independent set of questions. Although automated evaluation enabled large-scale analysis, results should be interpreted with caution, even as model rankings were preserved relative to human evaluation. The automated evaluator showed moderate agreement with human judgments (<italic>F</italic><sub>1</sub>-score=0.57), with corresponding false positive and false negative rates that introduce noise into correctness estimates. However, such noise has always been inherent to automated evaluation [<xref ref-type="bibr" rid="ref28">28</xref>-<xref ref-type="bibr" rid="ref30">30</xref>] and is mitigated by the large sample size, which improves the stability of relative comparisons despite reduced precision at the individual-turn level.</p><p>Third, this study did not explore the relationship between question difficulty and response correctness or harmfulness. Fourth, diversity measures like type-token ratio and entropy quantify surface-level lexical variety but do not necessarily reflect semantic or pragmatic diversity. In addition, patient personas were defined generically and did not explicitly incorporate comorbidities or medication use, limiting assessment of model performance in more clinically complex scenarios. We did not have access to a corpus of messages or phone call transcripts between patients and providers as these are privacy-protected; consequently, we were not able to quantitatively compare the synthetic conversations with real ones. Finally, because the colonoscopy preparation instructions used in prompts are publicly available, some models may have encountered them during pretraining. As a result, part of the observed performance could be inflated and reflect memorization rather than strictly reasoning from the prompts.</p></sec><sec id="s4-6"><title>Conclusions</title><p>Taken together, our results demonstrate that LLMs approach but do not yet achieve adequate performance in this task, given the number of harmful errors. Human and automatic evaluations together provide a nuanced understanding of model behavior, balancing interpretability and scalability. Prompt-based automatic filtering improves performance only for open models and does not fully prevent harmful errors, suggesting that the practical benefits of such straightforward methods remain limited. Future work should explore a variety of approaches for reducing the number of harmful errors, such as improving evaluator models through self-training [<xref ref-type="bibr" rid="ref31">31</xref>] or fine-tuning generator models to improve response quality [<xref ref-type="bibr" rid="ref32">32</xref>-<xref ref-type="bibr" rid="ref35">35</xref>]. Another promising direction is calibrating an evaluator model&#x2019;s confidence: overly cautious models risk unnecessary deferrals, while overconfident ones can allow harmful misinformation to pass through. Future research should aim to develop adaptive systems capable of calibrating their confidence based on the context and risk level of a patient&#x2019;s question. Future safety measures could also investigate rule-based medical constraint checking [<xref ref-type="bibr" rid="ref11">11</xref>]. Finally, testing on real patient queries will be necessary to validate model evaluation and help align models more closely with real patient expectations.</p></sec></sec></body><back><ack><p>We thank Arkobrato Gupta for his guidance and feedback on the statistical analysis. The authors used ChatGPT for language polishing but fully reviewed the content and take full responsibility for the manuscript.</p></ack><notes><sec><title>Funding</title><p>This research received no external funding.</p></sec><sec><title>Data Availability</title><p>The synthetic patient-AI Coach dialogues generated and evaluated in this study will be made publicly available on GitHub [<xref ref-type="bibr" rid="ref36">36</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>TK and MW conceived and designed the study. TK conducted the data generation, evaluation, and analysis, and drafted the manuscript under the supervision and guidance of MW. SC, KG, and IM contributed to dialogue annotation and provided feedback on the annotation guidelines. AP and EF-L contributed through prior discussions and general support. All authors reviewed and approved the final version of the manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CRAFT-MD</term><def><p>Conversational Reasoning Assessment Framework for Testing in Medicine</p></def></def-item><def-item><term id="abb2">EHR</term><def><p>electronic health records</p></def></def-item><def-item><term id="abb3">FAQ</term><def><p>frequently asked question</p></def></def-item><def-item><term id="abb4">GRADE</term><def><p>Grading of Recommendations Assessment, Development, and Evaluation</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">NSAID</term><def><p>nonsteroidal anti-inflammatory drug</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>Colorectal cancer statistics</article-title><source>Centers for Disease Control (CDC)</source><access-date>2026-03-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/colorectal-cancer/statistics/index.html">https://www.cdc.gov/colorectal-cancer/statistics/index.html</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Richardson</surname><given-names>LC</given-names> </name><name name-style="western"><surname>King</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>CC</given-names> </name><name name-style="western"><surname>Richards</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Dowling</surname><given-names>NF</given-names> </name><name name-style="western"><surname>Coleman King</surname><given-names>S</given-names> </name></person-group><article-title>Adults who have never been screened for colorectal cancer, Behavioral Risk Factor Surveillance System, 2012 and 2020</article-title><source>Prev Chronic Dis</source><year>2022</year><month>04</month><day>21</day><volume>19</volume><fpage>E21</fpage><pub-id pub-id-type="doi">10.5888/pcd19.220001</pub-id><pub-id pub-id-type="medline">35446758</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van der Zander</surname><given-names>QEW</given-names> </name><name name-style="western"><surname>Reumkens</surname><given-names>A</given-names> </name><name name-style="western"><surname>van de Valk</surname><given-names>B</given-names> </name><name name-style="western"><surname>Winkens</surname><given-names>B</given-names> </name><name name-style="western"><surname>Masclee</surname><given-names>AAM</given-names> </name><name name-style="western"><surname>de Ridder</surname><given-names>RJJ</given-names> </name></person-group><article-title>Effects of a personalized smartphone app on bowel preparation quality: randomized controlled trial</article-title><source>JMIR mHealth uHealth</source><year>2021</year><month>08</month><day>19</day><volume>9</volume><issue>8</issue><fpage>e26703</fpage><pub-id pub-id-type="doi">10.2196/26703</pub-id><pub-id pub-id-type="medline">34420924</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mahmud</surname><given-names>N</given-names> </name><name name-style="western"><surname>Doshi</surname><given-names>SD</given-names> </name><name name-style="western"><surname>Coniglio</surname><given-names>MS</given-names> </name><etal/></person-group><article-title>An automated text message navigation program improves the show rate for outpatient colonoscopy</article-title><source>Health Educ Behav</source><year>2019</year><month>12</month><volume>46</volume><issue>6</issue><fpage>942</fpage><lpage>946</lpage><pub-id pub-id-type="doi">10.1177/1090198119869964</pub-id><pub-id pub-id-type="medline">31431077</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sharara</surname><given-names>AI</given-names> </name><name name-style="western"><surname>Chalhoub</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Beydoun</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A customized mobile application in colonoscopy preparation: a randomized controlled trial</article-title><source>Clin Transl Gastroenterol</source><year>2017</year><month>01</month><day>5</day><volume>8</volume><issue>1</issue><fpage>e211</fpage><pub-id pub-id-type="doi">10.1038/ctg.2016.65</pub-id><pub-id pub-id-type="medline">28055031</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mahmud</surname><given-names>N</given-names> </name><name name-style="western"><surname>Asch</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Sung</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Effect of text messaging on bowel preparation and appointment attendance for outpatient colonoscopy: a randomized clinical trial</article-title><source>JAMA Netw Open</source><year>2021</year><month>01</month><day>4</day><volume>4</volume><issue>1</issue><fpage>e2034553</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2020.34553</pub-id><pub-id pub-id-type="medline">33492374</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Clancy</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Dominitz</surname><given-names>JA</given-names> </name></person-group><article-title>Texting to improve colonoscopy preparation and adherence needs more study</article-title><source>JAMA Netw Open</source><year>2021</year><month>01</month><day>4</day><volume>4</volume><issue>1</issue><fpage>e2035720</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2020.35720</pub-id><pub-id pub-id-type="medline">33492371</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gottweis</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="medline">39779926</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Arora</surname><given-names>RK</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hicks</surname><given-names>RS</given-names> </name><etal/></person-group><article-title>HealthBench: evaluating large language models towards improved human health</article-title><source>arXiv</source><comment>Preprint posted online on  May 13, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.08775</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Oufattole</surname><given-names>N</given-names> </name><name name-style="western"><surname>Weng</surname><given-names>WH</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Szolovits</surname><given-names>P</given-names> </name></person-group><article-title>What disease does this patient have? A large-scale open domain question answering dataset from medical exams</article-title><source>Appl Sci</source><year>2021</year><volume>11</volume><issue>14</issue><fpage>6421</fpage><pub-id pub-id-type="doi">10.3390/app11146421</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Arya</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bloomquist</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chakraborty</surname><given-names>S</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Stoyanchev</surname><given-names>S</given-names> </name><name name-style="western"><surname>Joty</surname><given-names>S</given-names> </name><name name-style="western"><surname>Schlangen</surname><given-names>D</given-names> </name><name name-style="western"><surname>Dusek</surname><given-names>O</given-names> </name><name name-style="western"><surname>Kennington</surname><given-names>C</given-names> </name><name name-style="western"><surname>Alikhani</surname><given-names>M</given-names> </name></person-group><article-title>Bootstrapping a conversational guide for colonoscopy prep</article-title><conf-name>24th Meeting of the Special Interest Group on Discourse and Dialogue</conf-name><conf-date>Sep 11-15, 2023</conf-date><conf-loc>Prague, Czechia</conf-loc><fpage>413</fpage><lpage>420</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.sigdial-1.38</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lan</surname><given-names>W</given-names> </name><etal/></person-group><article-title>An AI dietitian for type 2 diabetes mellitus management based on large language and image recognition models: preclinical concept validation study</article-title><source>J Med Internet Res</source><year>2023</year><month>11</month><day>9</day><volume>25</volume><fpage>e51300</fpage><pub-id pub-id-type="doi">10.2196/51300</pub-id><pub-id pub-id-type="medline">37943581</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sezgin</surname><given-names>E</given-names> </name><name name-style="western"><surname>Chekeni</surname><given-names>F</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Keim</surname><given-names>S</given-names> </name></person-group><article-title>Clinical accuracy of large language models and Google Search responses to postpartum depression questions: cross-sectional study</article-title><source>J Med Internet Res</source><year>2023</year><month>09</month><day>11</day><volume>25</volume><fpage>e49240</fpage><pub-id pub-id-type="doi">10.2196/49240</pub-id><pub-id pub-id-type="medline">37695668</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yalamanchili</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sengupta</surname><given-names>B</given-names> </name><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Quality of large language model responses to radiation oncology patient care questions</article-title><source>JAMA Netw Open</source><year>2024</year><month>04</month><day>1</day><volume>7</volume><issue>4</issue><fpage>e244630</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.4630</pub-id><pub-id pub-id-type="medline">38564215</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Guevara</surname><given-names>M</given-names> </name><name name-style="western"><surname>Moningi</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The effect of using a large language model to respond to patient messages</article-title><source>Lancet Digit Health</source><year>2024</year><month>06</month><volume>6</volume><issue>6</issue><fpage>e379</fpage><lpage>e381</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(24)00060-8</pub-id><pub-id pub-id-type="medline">38664108</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ayers</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Poliak</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dredze</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Comparing physician and artificial intelligence chatbot responses to patient questions posted to a public social media forum</article-title><source>JAMA Intern Med</source><year>2023</year><month>06</month><day>1</day><volume>183</volume><issue>6</issue><fpage>589</fpage><lpage>596</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2023.1838</pub-id><pub-id pub-id-type="medline">37115527</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Rezaei</surname><given-names>SJ</given-names> </name><etal/></person-group><article-title>Perspectives on artificial intelligence-generated responses to patient messages</article-title><source>JAMA Netw Open</source><year>2024</year><month>10</month><day>1</day><volume>7</volume><issue>10</issue><fpage>e2438535</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.38535</pub-id><pub-id pub-id-type="medline">39412810</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goodman</surname><given-names>RS</given-names> </name><name name-style="western"><surname>Patrinely</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Stone</surname><given-names>CA</given-names>  <suffix>Jr</suffix></name><etal/></person-group><article-title>Accuracy and reliability of chatbot responses to physician questions</article-title><source>JAMA Netw Open</source><year>2023</year><month>10</month><day>2</day><volume>6</volume><issue>10</issue><fpage>e2336483</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2023.36483</pub-id><pub-id pub-id-type="medline">37782499</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johri</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jeong</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tran</surname><given-names>BA</given-names> </name><etal/></person-group><article-title>An evaluation framework for clinical use of large language models in patient interaction tasks</article-title><source>Nat Med</source><year>2025</year><month>01</month><volume>31</volume><issue>1</issue><fpage>77</fpage><lpage>86</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03328-5</pub-id><pub-id pub-id-type="medline">39747685</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>M</given-names> </name></person-group><article-title>Synthetic data generation with large language models for text classification: potential and limitations</article-title><conf-name>2023 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Dec 6-10, 2023</conf-date><conf-loc>Singapore</conf-loc><fpage>10443</fpage><lpage>10461</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.647</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Shin</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Rahmani</surname><given-names>AM</given-names> </name></person-group><article-title>HealthQ: Unveiling questioning capabilities of LLM chains in healthcare conversations</article-title><source>Smart Health</source><year>2025</year><month>06</month><volume>36</volume><fpage>100570</fpage><pub-id pub-id-type="doi">10.1016/j.smhl.2025.100570</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Schuurmans</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Chain-of-thought prompting elicits reasoning in large language models</article-title><conf-name>Advances in Neural Information Processing Systems 35</conf-name><conf-date>Nov 28 to Dec 9, 2022</conf-date><conf-loc>New Orleans, Louisiana, USA</conf-loc><fpage>24824</fpage><lpage>24837</lpage><pub-id pub-id-type="doi">10.52202/068431-1800</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="web"><article-title>1 day bowel prep with Miralax and Dulcolax</article-title><source>The Ohio State University Wexner Medical Center</source><access-date>2025-06-06</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.healthwise.net/osumychart/Content/StdDocument.aspx?DOCHWID=custom.hs0146">https://www.healthwise.net/osumychart/Content/StdDocument.aspx?DOCHWID=custom.hs0146</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chong</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Verbalized sampling: how to mitigate mode collapse and unlock LLM diversity</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 1, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2510.01171</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>G</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>W</given-names> </name></person-group><article-title>Pride and prejudice: LLM amplifies self-bias in self-refinement</article-title><conf-name>62nd Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>Aug 11-16, 2024</conf-date><conf-loc>Bangkok, Thailand</conf-loc><fpage>15474</fpage><lpage>15492</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.acl-long.826</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="web"><person-group person-group-type="author"><collab>ATLAS.ti Team</collab></person-group><article-title>Krippendorff&#x2019;s alpha: sample size and decision rules</article-title><source>ATLAS.ti 26 Windows User Manual</source><year>2025</year><access-date>2026-07-07</access-date><publisher-name>Lumivero</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://manuals.atlasti.com/Win/en/manual/ICA/ICASampleSizeAndDecisionRules.html">https://manuals.atlasti.com/Win/en/manual/ICA/ICASampleSizeAndDecisionRules.html</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Gwet</surname><given-names>KL</given-names> </name></person-group><source>Handbook of Inter-Rater Reliability, 4th Edition: The Definitive Guide to Measuring the Extent of Agreement Among Raters</source><year>2014</year><edition>4</edition><publisher-name>Advanced Analytics</publisher-name><pub-id pub-id-type="other">9780970806284</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>D</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>L</given-names> </name><etal/></person-group><article-title>From generation to judgment: opportunities and challenges of LLM-as-a-judge</article-title><conf-name>2025 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 4-9, 2025</conf-date><conf-loc>Suzhou, China</conf-loc><fpage>2757</fpage><lpage>2791</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.emnlp-main.138</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Papineni</surname><given-names>K</given-names> </name><name name-style="western"><surname>Roukos</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ward</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>WJ</given-names> </name></person-group><article-title>BLEU: a method for automatic evaluation of machine translation</article-title><conf-name>40th Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>Jul 7-12, 2002</conf-date><fpage>311</fpage><lpage>318</lpage><pub-id pub-id-type="doi">10.3115/1073083.1073135</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reiter</surname><given-names>E</given-names> </name></person-group><article-title>A structured review of the validity of BLEU</article-title><source>Comput Linguist</source><year>2018</year><month>09</month><volume>44</volume><issue>3</issue><fpage>393</fpage><lpage>401</lpage><pub-id pub-id-type="doi">10.1162/coli_a_00322</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kulikov</surname><given-names>I</given-names> </name><name name-style="western"><surname>Golovneva</surname><given-names>O</given-names> </name><etal/></person-group><article-title>Self-taught evaluators</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 5, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2408.02666</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>A</given-names> </name><name name-style="western"><surname>White</surname><given-names>M</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Winning big with small models: knowledge distillation vs. self-training for reducing hallucination in QA agents</article-title><access-date>2026-07-28</access-date><conf-name>Fourth Workshop on Generation, Evaluation and Metrics (GEM)</conf-name><fpage>705</fpage><lpage>727</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.gem-1.62.pdf">https://aclanthology.org/2025.gem-1.62.pdf</ext-link></comment></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gulcehre</surname><given-names>C</given-names> </name><name name-style="western"><surname>Paine</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Srinivasan</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Reinforced self-training (ReST) for language modeling</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 17, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2308.08998</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Eisenach</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kakade</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Foster</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ghai</surname><given-names>U</given-names> </name></person-group><article-title>Mind the gap: examining the self-improvement capabilities of large language models</article-title><access-date>2026-07-07</access-date><conf-name>Proceedings of the 13th International Conference on Learning Representations</conf-name><conf-date>Apr 24-28, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.iclr.cc/paper_files/paper/2025/file/63943ee9fe347f3d95892cf87d9a42e6-Paper-Conference.pdf">https://proceedings.iclr.cc/paper_files/paper/2025/file/63943ee9fe347f3d95892cf87d9a42e6-Paper-Conference.pdf</ext-link></comment></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Eisenstein</surname><given-names>J</given-names> </name><name name-style="western"><surname>Aghajani</surname><given-names>R</given-names> </name><name name-style="western"><surname>Fisch</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Don&#x2019;t lie to your friends: learning what you know from collaborative self-play</article-title><access-date>2026-07-07</access-date><conf-name>1st Conference on Language Modeling (COLM 2025)</conf-name><conf-date>Jul 8-11, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=2vDJiGUfhV">https://openreview.net/pdf?id=2vDJiGUfhV</ext-link></comment></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="web"><article-title>Evaluating LLMs for colonoscopy preparation assistance</article-title><source>GitHub</source><year>2025</year><access-date>2025-11-20</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/sirimott/prep-coach-llms">https://github.com/sirimott/prep-coach-llms</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Prompt examples.</p><media xlink:href="ai_v5i1e88581_app1.docx" xlink:title="DOCX File, 4022 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2 </label><p>Prep instructions.</p><media xlink:href="ai_v5i1e88581_app2.docx" xlink:title="DOCX File, 4022 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3 </label><p>Error examples.</p><media xlink:href="ai_v5i1e88581_app3.docx" xlink:title="DOCX File, 4025 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4 </label><p>Most frequent tokens.</p><media xlink:href="ai_v5i1e88581_app4.docx" xlink:title="DOCX File, 46 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5 </label><p>Baseline correctness.</p><media xlink:href="ai_v5i1e88581_app5.docx" xlink:title="DOCX File, 4020 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6 </label><p>Correctness by temporal window.</p><media xlink:href="ai_v5i1e88581_app6.docx" xlink:title="DOCX File, 320 KB"/></supplementary-material></app-group></back></article>