<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Ment Health</journal-id><journal-id journal-id-type="publisher-id">mental</journal-id><journal-id journal-id-type="index">16</journal-id><journal-title>JMIR Mental Health</journal-title><abbrev-journal-title>JMIR Ment Health</abbrev-journal-title><issn pub-type="epub">2368-7959</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v13i1e94781</article-id><article-id pub-id-type="doi">10.2196/94781</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Large Language Model&#x2013;Based Behavioral Activation Chatbot for Young People With Depression Using Artificial Users and Clinical Experts: Mixed Methods Evaluation</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Kuhlmeier</surname><given-names>Florian Onur</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hanschmann</surname><given-names>Leon</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Rabe</surname><given-names>Melina</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>L&#x00FC;ttke</surname><given-names>Stefan</given-names></name><degrees>Dipl-Psych</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Brakemeier</surname><given-names>Eva-Lotta</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Maedche</surname><given-names>Alexander</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Institute for Information Systems (WIN), Karlsruhe Institute of Technology</institution><addr-line>Kaiserstra&#x00DF;e 89-93</addr-line><addr-line>Karlsruhe</addr-line><addr-line>Baden-Wurttemberg</addr-line><country>Germany</country></aff><aff id="aff2"><institution>Department of Clinical Psychology and Psychotherapy, Universit&#x00E4;t Greifswald</institution><addr-line>Greifswald</addr-line><addr-line>Mecklenburg-Vorpommern</addr-line><country>Germany</country></aff><aff id="aff3"><institution>Chair of Clinical Child and Adolescent Psychology and Psychotherapy, Department of Psychology, Saarland University</institution><addr-line>Saarbr&#x00FC;cken</addr-line><country>Germany</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Knapp</surname><given-names>Ashley</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Jabir</surname><given-names>Ahmad</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Chakit</surname><given-names>Miloud</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Khanna</surname><given-names>Varada Vivek</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Hu</surname><given-names>Yihan</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Florian Onur Kuhlmeier, PhD, Institute for Information Systems (WIN), Karlsruhe Institute of Technology, Kaiserstra&#x00DF;e 89-93, Karlsruhe, Baden-Wurttemberg, 76133, Germany, 49 721 608-48380; <email>florian.kuhlmeier@kit.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>1</day><month>9</month><year>2026</year></pub-date><volume>13</volume><elocation-id>e94781</elocation-id><history><date date-type="received"><day>06</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>04</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>12</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Florian Onur Kuhlmeier, Leon Hanschmann, Melina Rabe, Stefan L&#x00FC;ttke, Eva-Lotta Brakemeier, Alexander Maedche. Originally published in JMIR Mental Health (<ext-link ext-link-type="uri" xlink:href="https://mental.jmir.org">https://mental.jmir.org</ext-link>), 1.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Mental Health, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mental.jmir.org/">https://mental.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mental.jmir.org/2026/1/e94781"/><abstract><sec><title>Background</title><p>Mental health chatbots are increasingly used to support people with depressive symptoms, and large language models make these systems more flexible than rule-based chatbots. However, it remains unclear how well large language model&#x2013;based chatbots deliver structured psychological interventions.</p></sec><sec><title>Objective</title><p>This study examined how well a GPT-4o&#x2013;based chatbot delivered a behavioral activation intervention for young people with depression using sessions with artificial users and clinical expert assessment. It also identified limitations and potential refinements.</p></sec><sec sec-type="methods"><title>Methods</title><p>We implemented a GPT-4o (gpt-4o-2024-08-06; OpenAI)&#x2013;based chatbot using a structured system prompt to deliver a single-session behavioral activation intervention for people with depression aged 14 to 29 years. We generated 48 sessions with GPT-4o&#x2013;based artificial users derived from clinical vignettes varying across 7 characteristics. Ten clinical experts, either licensed psychotherapists or advanced psychotherapy trainees, independently assessed the sessions using the 14-item Quality of Behavioral Activation Scale (Q-BAS), rated from 0 to 6, supplemented by rating therapeutic capabilities, artificial user authenticity and difficulty, and qualitative feedback.</p></sec><sec sec-type="results"><title>Results</title><p>The chatbot completed all 7 intervention phases in every session. The mean holistic session quality rating was 3.94 (SD 1.23), and the mean Q-BAS rating was 4.03 (SD 1.18). Thirteen of 14 Q-BAS components exceeded the satisfactory threshold of 3. Ratings were highest for mood assessment (mean 5.42, SD 1.09) and activity planning (mean 4.98, SD 1.41) and lowest for explaining positive reinforcement (mean 2.92, SD 2.30) and supporting activity-mood monitoring (mean 3.02, SD 2.04). Therapeutic capability ratings were highest for message safety (mean 5.90, SD 0.37), message clarity (mean 5.56, SD 0.77), and objective, nonjudgmental communication (mean 5.17, SD 1.04) and lowest for therapeutic rapport (mean 4.12, SD 1.45) and natural conversation flow (mean 4.25, SD 1.42). Artificial users were rated below the scale midpoint for authenticity (mean 2.75, SD 1.41) and difficulty (mean 1.23, SD 1.46). Clinical experts described the chatbot as structured, clear, and safe but identified insufficient clinical reasoning as the main limitation, particularly in evaluating the therapeutic suitability and feasibility of activities, barriers, solution strategies, and rewards. Artificial users were often highly compliant, especially when identifying positive activities.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In expert-rated sessions with artificial users, the chatbot delivered the behavioral activation intervention as intended and performed strongest on procedural components. It performed less well on positive reinforcement and activity-mood monitoring, indicating refinement needs in clinical reasoning, follow-up questioning, and evaluating whether proposed activities, plans, barriers, solution strategies, and rewards are therapeutically appropriate and feasible. The findings identify targets for improvement before testing with human users, while the artificial user design and expert ratings limit conclusions about real therapeutic interactions.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>mental health chatbots</kwd><kwd>behavioral activation</kwd><kwd>depression</kwd><kwd>clinical fidelity</kwd><kwd>artificial users</kwd><kwd>prompt engineering</kwd><kwd>young people</kwd><kwd>digital mental health interventions</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Mental health chatbots are increasingly used as scalable and accessible tools to support people with depressive symptoms [<xref ref-type="bibr" rid="ref1">1</xref>]. Rule-based chatbots, such as Woebot [<xref ref-type="bibr" rid="ref2">2</xref>] and Wysa [<xref ref-type="bibr" rid="ref3">3</xref>], have been shown to reduce depressive symptoms, but their scripted messages and predefined response paths can make interactions rigid and repetitive, which can reduce responsiveness to individual user needs and contribute to insufficient engagement or symptom improvement [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Large language models (LLMs) can address some of these limitations because they can generate flexible and context-sensitive responses. However, this same flexibility creates an evaluation problem. LLM-based chatbots can respond inconsistently or harmfully [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>], and fluent therapeutic language does not necessarily mean that an intervention is delivered with clinical fidelity. Therefore, the key question is not whether they can sound therapeutic but whether they can deliver a structured psychological intervention as intended. In psychotherapy training and clinical studies, clinical experts commonly use fidelity instruments to evaluate whether psychotherapists deliver intervention components as intended ([<xref ref-type="bibr" rid="ref9">9</xref>] and Dimidjian S, Hubley S, Martell C, Herman-Dunn A, Dobson K. The Quality of Behavioral Activation Scale [Q-BAS], unpublished instrument, 2012, University of Colorado Boulder). LLM-based mental health chatbots have rarely been assessed using comparable process measures. Existing evaluations often examine single-turn responses, general response quality, usability, and downstream outcomes [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref14">14</xref>]. These approaches provide useful evidence but do not show whether a chatbot can sustain the components of an intervention across a full session. Recent work on retrieval-grounded evaluation for conversational LLM-based risk assessment makes a related point: clinically sensitive LLM systems need evaluation methods that go beyond aggregate response quality and examine clinical fidelity, safety, and behavior across user subgroups [<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>Behavioral activation offers a useful setting for evaluating LLM-based mental health chatbots before conducting studies involving human users. It is one of the most empirically supported psychological treatments for depression in young people [<xref ref-type="bibr" rid="ref16">16</xref>], and its structured protocol makes it possible to assess whether the core intervention components are delivered. Testing an unevaluated LLM-based chatbot with young people experiencing depressive symptoms raises practical and ethical concerns. Artificial users can address this problem by generating standardized, repeatable, and clinically varied simulated interactions before human user studies [<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref21">21</xref>]. Therefore, they provide a controlled evaluation layer for identifying intervention-delivery weaknesses before the chatbot is studied with human users.</p><p>We assessed a prompt-engineered GPT-4o (gpt-4o-2024-08-06; OpenAI) behavioral activation chatbot for young people with depression. The primary aim was to examine how well the chatbot delivered the core components of a structured behavioral activation protocol and where its delivery required refinement before studies with human users. Secondary analyses examined broader therapeutic capabilities, authenticity and difficulty of the artificial users, and associations between artificial user characteristics and ratings.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>We used a mixed methods design to evaluate a prompt-engineered behavioral activation chatbot implemented by our research team. The evaluation combined artificial user session generation with independent clinical expert assessment. Artificial users generated complete therapeutic session transcripts across diverse clinical presentations, and clinical experts rated these transcripts using a validated fidelity instrument and provided open-ended feedback.</p></sec><sec id="s2-2"><title>Behavioral Activation Chatbot</title><p>We implemented a GPT-4o&#x2013;powered behavioral activation chatbot that delivered a single-session protocol [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>] for young people aged 14 to 29 years with depressive symptoms. The chatbot was built based on the rule-based chatbot Cady [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>] and used a structured system prompt to guide interactions through the intervention protocol. The development process involved close collaboration with clinical experts who translated the rule-based script into a structured prompt format and crafted illustrative examples. The system prompt underwent iterative refinement, with development team members role-playing as users and clinical experts evaluating the chatbot&#x2019;s performance. Initial testing revealed that increased conversational flexibility reduced protocol adherence and produced inconsistent responses, leading us to prioritize a linear progression through all 7 phases.</p><p>The final German-language system prompt comprised 6 hierarchical components, which are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. The prompt was designed to guide a complete behavioral activation session while maintaining a conversational exchange. Format instructions used phase markers, such as [phase 1] and [phase 2], to track progress through the session and a [STOP] marker to signal completion. The task instructions guided the chatbot through a 7-phase behavioral activation protocol aimed at creating a personalized activity plan for the user. Phase-specific instructions defined the goal and completion criteria for each phase and included contrastive good and bad example dialogues to reduce common errors through in-context learning. Safety and boundary constraints included a 30-word message limit, referral to emergency services when suicidal ideation was disclosed, and polite redirection of off-topic requests from users. To reduce instruction drift during longer conversations, the format instructions were repeated at the beginning and the end of the prompt. The full prompt is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Prompt architecture.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Component</td><td align="left" valign="bottom">Purpose</td><td align="left" valign="bottom">Content summary</td></tr></thead><tbody><tr><td align="left" valign="top">Format instructions</td><td align="left" valign="top">Phase transition control and session structure</td><td align="left" valign="top">Explicit markers that track session progress (eg, [phase 1]) and enforce sequential phase completion</td></tr><tr><td align="left" valign="top">Identity</td><td align="left" valign="top">Role and persona definition</td><td align="left" valign="top">Chatbot for young people experiencing depressive symptoms, featuring an empathetic, activating, encouraging, humorous, and curious personality</td></tr><tr><td align="left" valign="top">Constraints</td><td align="left" valign="top">Communication guidelines and safety protocols</td><td align="left" valign="top">Includes a 30-word message limit, suicide or emergency protocol with crisis referral, and role boundaries that politely decline off-topic requests</td></tr><tr><td align="left" valign="top">Task</td><td align="left" valign="top">Overall therapeutic goal</td><td align="left" valign="top">Instructs the chatbot to guide the user through a 7-phase behavioral activation session with the key objective of collaboratively creating a personal activity plan</td></tr><tr><td align="left" valign="top">Phase-specific instructions</td><td align="left" valign="top">Detailed phase procedures</td><td align="left" valign="top">Seven phases (introduction, psychoeducation, finding activities, planning activities, problem solving, positive reinforcements, and closing), each with specific goals, completion criteria, and good or bad example sessions</td></tr><tr><td align="left" valign="top">Complete session example</td><td align="left" valign="top">Comprehensive session model</td><td align="left" valign="top">A full multiturn session demonstrating all 7 phases with natural pacing and smooth transitions</td></tr></tbody></table></table-wrap><p>The chatbot was implemented using GPT-4o via the OpenAI API, with the temperature set to 1 to balance response consistency and variety because, in our initial tests, lower values produced overly repetitive replies.</p></sec><sec id="s2-3"><title>Artificial Users and Session Generation</title><p>We developed artificial users based on patient vignettes, which are concise clinical case descriptions commonly used in psychotherapy research and training [<xref ref-type="bibr" rid="ref26">26</xref>]. The vignettes were selected from psychotherapy training materials used at the outpatient clinic of the University of Greifswald. Each vignette described a young person aged 14 to 29 years with depression, including symptoms and psychosocial circumstances. We selected 4 base vignettes to cover different demographic backgrounds and symptom presentations.</p><p>To capture variation among potential users, we enriched the base vignettes with 7 characteristics identified through a literature review and consultations with clinical experts. These characteristics included depression severity, age, gender, willingness to disclose personal information, openness to chatbot suggestions, conversational dominance, and attitudes toward mental health chatbots. Each characteristic was implemented using specific text expressions added to the base vignettes. <xref ref-type="table" rid="table2">Table 2</xref> summarizes the characteristics, variation levels, and rationale for the selection. The complete vignettes and variation expressions are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Artificial user characteristics and rationale for selection.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Variations</td><td align="left" valign="bottom">Rationale for selection</td></tr></thead><tbody><tr><td align="left" valign="top">Depression severity</td><td align="left" valign="top">Mild, moderate, severe</td><td align="left" valign="top">Higher severity increases interest and willingness to adopt digital mental health interventions but hampers actual engagement due to depressive symptoms, low mood, and fatigue that inhibit motivation and ability to use interventions [<xref ref-type="bibr" rid="ref27">27</xref>]</td></tr><tr><td align="left" valign="top">Age</td><td align="left" valign="top">14&#x2010;17, 18&#x2010;25, 26&#x2010;29</td><td align="left" valign="top">Younger users exhibit different depressive symptoms [<xref ref-type="bibr" rid="ref28">28</xref>]</td></tr><tr><td align="left" valign="top">Gender</td><td align="left" valign="top">Male, female, nonbinary</td><td align="left" valign="top">Women are more likely to engage with digital mental health interventions than men [<xref ref-type="bibr" rid="ref27">27</xref>]</td></tr><tr><td align="left" valign="top">Willingness to disclose personal information</td><td align="left" valign="top">High, low</td><td align="left" valign="top">Privacy concerns and confidentiality fears create barriers to engagement and information disclosure [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]</td></tr><tr><td align="left" valign="top">Openness to chatbot suggestions</td><td align="left" valign="top">High, low</td><td align="left" valign="top">Preexisting beliefs about digital intervention effectiveness affect engagement [<xref ref-type="bibr" rid="ref27">27</xref>]</td></tr><tr><td align="left" valign="top">Dominance</td><td align="left" valign="top">High, low</td><td align="left" valign="top">Dominance affects conversations between users and chatbots [<xref ref-type="bibr" rid="ref30">30</xref>]</td></tr><tr><td align="left" valign="top">Attitudes toward mental health chatbots</td><td align="left" valign="top">Positive, negative</td><td align="left" valign="top">Negative attitudes and the &#x201C;humans need humans&#x201D; preference for in-person therapy create barriers to digital uptake [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]</td></tr></tbody></table></table-wrap><p>To verify that the artificial users matched the intended depression severity levels, each artificial user completed the Patient Health Questionnaire-9 (PHQ-9) in a separate prompt before interacting with the chatbot (<xref ref-type="table" rid="table3">Table 3</xref>). The prompt used only the artificial user description as the context. Artificial users with scores outside the intended range were excluded and resampled. We used the following severity ranges, adapted from the standard PHQ-9 categories, to fit our 3-level classification: mild, 5 to 9; moderate, 10 to 19; and severe, 20 to 27.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Example artificial user.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Level</td><td align="left" valign="bottom">Description</td></tr></thead><tbody><tr><td align="left" valign="top">Base vignette (gender, age group, and depression severity)</td><td align="left" valign="top">Female, young, adult, and severe depression</td><td align="left" valign="top">I am Kira, 29 years old, and I am just hanging around in my flat. I have lost my job as a paralegal, and now everything is completely messed up. I constantly feel as if I am in a black hole. A relationship? Not a chance. My friends are getting married and having children, but I feel completely disconnected and isolated. In addition, my mother has developed Alzheimer disease. That completely knocks me out. My sleep rhythm no longer exists. I lie awake for hours and cannot fall asleep. If I doze off, I wake up again a few hours later and then lie awake until dawn. I often get up at 4 or 5 AM because there is no point anyway. Food? Forget it; I have no appetite at all. Dating has not been a thing for a long time. I just stay at home and do not feel like doing anything. I am permanently down and cannot concentrate on anything. I often ask myself what the point of all this is. At home, I constantly brood about my job loss and feel like a failure. Everything seems pointless to me. I lie awake at night, worrying that I will go completely broke. I have driven all my friends away. I feel totally worthless and have extreme feelings of guilt regarding everything. Sometimes, I can hardly move, and even showering is torture. I constantly think about what it would be like if I were no longer there. Sometimes, I really think about whether I should just end it.</td></tr><tr><td align="left" valign="top">Willingness to disclose information</td><td align="left" valign="top">High</td><td align="left" valign="top">I provide detailed answers to the chatbot&#x2019;s questions and willingly share specific examples from my life.</td></tr><tr><td align="left" valign="top">Openness to suggestions</td><td align="left" valign="top">High</td><td align="left" valign="top">I am receptive to the chatbot&#x2019;s suggestions and willingly try out its recommendations. When the chatbot proposes new approaches, I am eager to explore them and give them a fair chance.</td></tr><tr><td align="left" valign="top">Conversational dominance</td><td align="left" valign="top">High</td><td align="left" valign="top">I confidently steer the conversation by asking the chatbot specific questions and clearly formulating my expectations of the therapy.</td></tr><tr><td align="left" valign="top">Attitudes toward chatbot</td><td align="left" valign="top">Negative</td><td align="left" valign="top">I am critical of using chatbots. Instead, I would prefer to see a human therapist.</td></tr></tbody></table></table-wrap><p>From the verified pool, we drew a stratified random sample of 48 artificial users, with a roughly balanced representation across the 7 characteristics. The sample size was determined by the number of available clinical experts (n=10) and their available evaluation time (1&#x2010;2 h per clinical expert), which allowed each clinical expert to assess 3 to 6 sessions.</p><p>Similar to the chatbot, artificial users were implemented by prompting GPT-4o, with the temperature set to 1. During pilot testing, this setting provided the best balance between adherence to the artificial user description and the variety of responses. The artificial user prompts and generated interactions were implemented in German. Sessions began with a standardized welcome message from the behavioral activation chatbot and were generated using a Python script. A session ended when the chatbot sent the STOP] marker, completed all 7 phase markers, or reached a 100-turn cap. We verified session completion by screening the transcripts for these predefined stopping criteria.</p></sec><sec id="s2-4"><title>Expert Assessment</title><sec id="s2-4-1"><title>Participants</title><p>We recruited participants between August and September 2024 from licensed psychotherapists and advanced psychotherapy trainees affiliated with the outpatient clinic of the University of Greifswald, as well as from the research team&#x2019;s professional network. The inclusion criteria were as follows: (1) a master&#x2019;s degree in psychology, (2) licensure as a psychotherapist or enrollment in psychotherapist training from the second year onward, and (3) experience treating young people with depression using behavioral activation. Ten participants were recruited (mean age 30.1, SD 4.12 y; n=7, 70% female participants; mean clinical experience 3.75, SD 1.75 y). Two were licensed psychotherapists, and 8 were advanced psychotherapy trainees. Seven participants reported prior experience with digital mental health interventions. All participants were distinct from the psychotherapy experts involved in chatbot development. Each participant received &#x20AC;30 (&#x20AC;1=US $1.11 as of September 30, 2024) for 1 to 2 hours of participation.</p></sec><sec id="s2-4-2"><title>Study Procedure</title><p>The evaluation comprised 3 phases: first, the participants received information about the study objectives and procedures and provided written informed consent. Second, each participant independently assessed 3 to 6 complete chatbot sessions using LimeSurvey. Participants read each session in full before completing the questionnaire, and the session transcript remained accessible while they answered questions. Breaks were allowed within and between the sessions to reduce fatigue. Each session was evaluated by a single participant. Third, semistructured interviews explored participants&#x2019; perspectives on the sessions and the artificial user approach. Participants were informed before the questionnaire phase that the sessions involved a chatbot, but the use of artificial users was disclosed only during the interview phase to reduce bias in the fidelity assessment.</p></sec><sec id="s2-4-3"><title>Measures</title><sec id="s2-4-3-1"><title>Behavioral Activation Fidelity</title><p>We used the Quality of Behavioral Activation Scale (Q-BAS) (Dimidjian S, Hubley S, Martell C, Herman-Dunn A, Dobson K. The Quality of Behavioral Activation Scale [Q-BAS], unpublished instrument, 2012, University of Colorado Boulder), adapted for chatbot delivery, to assess the quality of behavioral activation delivery. The Q-BAS includes 14 items rated on a 7-point Likert scale, with higher scores indicating better delivery. The Q-BAS defines scores of 3 or higher as satisfactory delivery of behavioral activation components, and this threshold has been applied in studies of human therapists delivering behavioral activation in person and via teletherapy [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. Because this satisfactory threshold has not been validated for chatbot-delivered behavioral activation, we used it as a descriptive benchmark rather than as an indicator of clinical adequacy.</p></sec><sec id="s2-4-3-2"><title>Holistic Session Quality</title><p>A single item assessed the overall session quality: &#x201C;Overall, how would you rate the chatbot as a behavioral activation chatbot in this session?&#x201D; It was rated on a 7-point Likert scale, with higher scores indicating higher quality.</p></sec><sec id="s2-4-3-3"><title>Therapeutic Capabilities</title><p>Seven items adapted from the Thera-Turing Test [<xref ref-type="bibr" rid="ref33">33</xref>] assessed broader therapeutic capabilities: emotional validation and empathy, responsiveness to user concerns, therapeutic rapport, objectivity and nonjudgment, message clarity, natural conversation flow, and message safety. Each item used a 7-point agreement scale, with higher scores indicating stronger agreement that the chatbot demonstrated the respective capability.</p></sec><sec id="s2-4-3-4"><title>Artificial User Authenticity and Difficulty</title><p>Clinical experts rated each artificial user&#x2019;s perceived authenticity and the difficulty of conducting the session. Both items used a 7-point Likert scale, with higher scores indicating greater perceived authenticity and greater artificial-user difficulty, respectively.</p></sec><sec id="s2-4-3-5"><title>Qualitative Feedback</title><p>Open-ended questionnaire items asked participants to explain what the chatbot did well and what it could have done better overall and in each phase of the behavioral activation protocol.</p></sec></sec></sec><sec id="s2-5"><title>Ethical Considerations</title><p>The Institutional Review Board of the Karlsruhe Institute of Technology approved this study prior to data collection (reference number: A2024-095). Because the chatbot&#x2019;s clinical fidelity had not yet been evaluated, artificial-user testing was approved as an intermediate step before studies with human users. All participants provided written informed consent before participation. Rating data were pseudonymized and stored separately from personally identifiable information to protect participant privacy and confidentiality. Participants received &#x20AC;30 in compensation for 1 to 2 hours of participation.</p></sec><sec id="s2-6"><title>Data Analysis</title><p>Quantitative data were analyzed using R (version 4.3.1; R Foundation for Statistical Computing). All quantitative outcomes were based on 7-point Likert scales and are reported numerically on a 0 to 6 scale.</p><p>Q-BAS ratings were analyzed item-wise and session-wise. Component-wise analyses summarized the ratings for each of the 14 Q-BAS components across all 48 sessions. Session-wise analyses summarized the 14 component ratings within each session. We calculated descriptive statistics, compared component ratings with the predefined satisfactory delivery threshold of &#x2265;3, and computed the session-level Q-BAS mean as the average of the 14 component scores. To describe variation in Q-BAS ratings, we fitted a linear mixed-effects model with crossed random intercepts for sessions and intervention components using restricted maximum likelihood estimation. Variance components were extracted from the fitted model, and 95% profile-likelihood CIs were derived from the same model. Because the profile-likelihood CIs for random effects are estimated on the SD scale, the interval limits were squared to obtain CIs on the variance scale. In the primary analysis, each session was rated by a clinical expert. Therefore, session-level and rater-level variance could not be separated, and the session-level variance component was interpreted descriptively.</p><p>The therapeutic capability ratings were analyzed item-wise. Item-wise analyses summarized the ratings for each of the 7 therapeutic capabilities across all 48 sessions. We calculated descriptive statistics for each capability and analyzed the items separately because they capture distinct capabilities rather than a unified therapeutic capability construct.</p><p>For the interrater agreement sensitivity analysis, an additional licensed therapist rated a randomly selected subset of 18 of the 48 dialogues. These additional ratings were used only for agreement analyses and were not included in the primary results. We calculated the mean absolute difference and intraclass correlation coefficient (ICC [2,1] for the Q-BAS mean, and the median absolute difference and quadratic weighted &#x03BA; for the ordinal holistic session quality rating. As another sensitivity analysis, we compared ratings between licensed psychotherapists and psychotherapy trainees.</p><p>Exploratory hypothesis-generating analyses were conducted to examine whether artificial user characteristics were associated with Q-BAS ratings, therapeutic capability ratings, artificial user authenticity, and artificial-user difficulty. We used Wilcoxon rank-sum tests for 2-level artificial user characteristics and Kruskal-Wallis tests for 3-level artificial user characteristics. To account for multiple comparisons, <italic>P</italic> values were adjusted using the Benjamini-Hochberg (BH) false discovery rate procedure within 3 outcome domain families: Q-BAS outcomes, therapeutic capability ratings, and artificial user ratings. We report both raw and BH-adjusted <italic>P</italic> values.</p><p>Open-ended questionnaire responses and semistructured interviews were analyzed using qualitative content analysis [<xref ref-type="bibr" rid="ref34">34</xref>]. The 7 intervention phases and 14 Q-BAS components served as deductive categories, and additional categories were developed inductively from the data. One researcher coded the material and developed the category system through iterative review. The resulting categories and ambiguous coding decisions were discussed with a second researcher and refined as needed. We report category and code frequencies to make the analysis transparent and to indicate the salience of themes in the data.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Behavioral Activation Fidelity</title><p><xref ref-type="fig" rid="figure1">Figure 1</xref> shows the Q-BAS ratings and the holistic single-item session quality across the 48 evaluated sessions.</p><p>The chatbot received a mean rating of 3.94 (SD 1.23) on the holistic single-item rating of overall session quality. The average Q-BAS rating across the 14 behavioral activation components was 4.03 (SD 1.18). Thirteen of the 14 components exceeded the satisfactory threshold of 3 on average. Mood assessment received the highest rating (mean 5.42, SD 1.09), followed by planning activities (mean 4.98, SD 1.41). The weakest components were explaining positive reinforcement (mean 2.92, SD 2.30), which was the only component below the satisfactory threshold, and encouraging users to observe activity-mood connections (mean 3.02, SD 2.04). At the component level, satisfactory threshold showed a similar pattern: mood assessment was &#x2265;3 in 47 of 48 (98%) sessions, while explaining positive reinforcement was &#x2265;3 in 27 of 48 (56%) sessions. <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> provides the full satisfactory threshold rates across all sessions.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Quality of behavioral activation scale (Q-BAS) ratings by intervention component. The bars show the distribution of ratings across all 48 sessions, with the color indicating the rating. The components are ordered by intervention phase (P1-P7). Holistic single-item session quality is shown at the top. Means and SDs are reported for each component of the scale.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e94781_fig01.png"/></fig><p>In the variance decomposition analysis, 37% of the variance in Q-BAS scores was estimated at the session level (variance =1.27, 95% CI 0.82-2.02), 12% at the intervention-component level (variance 0.42, 95% CI 0.18-0.98), and 51% as the residual variance. In the repeated-rating subset (n=18), the mean absolute difference in session-level Q-BAS means was 0.68 (SD 0.59). The ICC (2,1) for absolute agreement was 0.55 (95% CI 0.15-0.80). For holistic session quality, the median absolute difference was 1 point (IQR 1), and the quadratic weighted &#x03BA; was 0.63 (95% CI &#x2013;0.07 to 0.85). Licensed psychotherapists rated Q-BAS means descriptively higher than trainees (mean 4.28, SD 0.82 vs mean 3.55, SD 1.21; Wilcoxon <italic>P</italic>=.23). Holistic ratings showed a similar descriptive pattern (mean 4.00, SD 0.87 vs mean 3.44, SD 1.33; <italic>P</italic>=.37). The full sensitivity analyses are reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>The clinical experts&#x2019; qualitative comments helped to contextualize the fidelity ratings. They highlighted the chatbot&#x2019;s structured session flow (n=7) and ability to validate users&#x2019; feelings (n=3) as strengths but also described some sessions as superficial or less detailed than typical therapy sessions (n=7). Across phases, they suggested deeper follow-up questions during mood assessment, more personalized psychoeducation, more guidance for resistant users, and stronger checks on whether activities, barriers, solution strategies, and rewards were suitable and feasible. Concrete examples included accepting an evening nap as an activity without checking whether it was appropriate, proposing one-sided reward options, and ending sessions without sufficient guidance on activity-mood monitoring.</p><p>Inspection of the session-level heatmap in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> identified 2 sessions with consistently low Q-BAS ratings across components. Content analysis suggested that both sessions were shaped by user steering patterns that disrupted the protocol delivery. In 1 session, the user adopted an information-seeking stance and asked broad self-help questions regarding motivation, stress management, and social skills. In another session, the user engaged in skeptical probing by repeatedly asking, &#x201C;What if that doesn&#x2019;t help?&#x201D; In both cases, the chatbot responded reactively, rather than redirecting the conversation toward the behavioral activation protocol. As a result, these sessions shifted toward unstructured question-and-answer exchanges, and the chatbot struggled to complete the behavioral activation session.</p></sec><sec id="s3-2"><title>Therapeutic Capabilities</title><p><xref ref-type="fig" rid="figure2">Figure 2</xref> provides an overview of how the clinical experts rated the therapeutic capabilities of the chatbot.</p><p>The ratings were above the scale midpoint for all therapeutic capabilities. Message safety received the highest mean rating (mean 5.90, SD 0.37), followed by message clarity (mean 5.56, SD 0.77) and objective, nonjudgmental communication (mean 5.17, SD 1.04). Of the 48 evaluated sessions, 44 (92%) received the maximum safety rating of 6, and the remaining sessions were rated 5 (n=3, 6%) or 4 (n=1, 2%). No session received a safety rating below 4. Artificial users with high depression severity contained suicidality. However, no session included an explicit disclosure of suicidal ideation, suicidal intent, or self-harm thoughts; therefore, the crisis protocol was not triggered. The details are reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Therapeutic capability ratings across 48 sessions. The bars show the distribution of ratings across all 48 sessions, with color indicating the rating. The means and SDs are reported for each capability.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e94781_fig02.png"/></fig><p>The ratings were still high but lower for responding appropriately to user concerns (mean 4.94, SD 1.08), validating feelings and showing empathy (mean 4.81, SD 1.00), natural conversation flow (mean 4.25, SD 1.42), and building therapeutic rapport (mean 4.12, SD 1.45). The largest SDs were observed for therapeutic rapport and natural conversation flow, indicating greater variability.</p><p>Qualitative comments were aligned with these ratings. Clinical experts highlighted the chatbot&#x2019;s safety, clear communication, and objective and nonjudgmental tone as strengths, and they did not raise safety concerns. They also identified areas for refinement: validation was sometimes too brief or generic, responses to user concerns were sometimes incomplete, enthusiastic expressions could feel exaggerated, some questions were suggestive, and transitions between psychoeducation and planning were sometimes abrupt. Several comments linked relational quality to concrete interaction behaviors, including stronger validation, clearer acknowledgment of users&#x2019; doubts or low motivation, and more individualized follow-ups.</p></sec><sec id="s3-3"><title>Artificial User Authenticity and Difficulty</title><p><xref ref-type="fig" rid="figure3">Figure 3</xref> summarizes the ratings for artificial user authenticity and difficulty.</p><p>Clinical experts rated artificial user authenticity as slightly below the scale midpoint (mean 2.75, SD 1.41; median 2.50, IQR 2-4) and artificial-user difficulty as low (mean 1.23, SD 1.46; median 1.00, IQR 0-2).</p><p>Qualitative comments clarified the main limitations of artificial users. Experts primarily criticized their high compliance with the chatbot&#x2019;s suggestions, especially when identifying positive activities. One expert noted that real patients often need more support at this point, saying, &#x201C;Usually it is first &#x2018;I don&#x2019;t know any activities&#x2019; or &#x2018;I don&#x2019;t remember any.&#x2019;&#x201D; In contrast, experts described the clinical background stories as plausible, including vignettes of worsening mental health after COVID-19.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Artificial user ratings were obtained across 48 sessions. The bars show the distribution of ratings across all 48 sessions, with color indicating the rating level. Means and SDs are reported for each rating.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e94781_fig03.png"/></fig><p>Artificial users with negative attitudes toward mental health chatbots were rated as more authentic than those with positive attitudes (mean 3.16, SD 1.52 vs mean 2.30, SD 1.15; Wilcoxon W=387.50, <italic>P</italic>=.04). Q-BAS ratings, message safety, and message clarity also varied by artificial user openness to chatbot suggestions and their willingness to disclose information. None of these associations remained significant after the BH correction. The complete results are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><p>Our findings suggest that the GPT-4o&#x2013;powered chatbot delivered a structured behavioral activation session effectively in this evaluation setting. Across 48 artificial user sessions, the chatbot completed all 7 intervention phases and received a mean Q-BAS rating of 4.03 (SD 1.18) on a 0 to 6 scale. Thirteen of the 14 behavioral activation components exceeded the Q-BAS satisfactory threshold. The strongest ratings were for components with explicit procedural goals, particularly mood assessment and activity planning. Overall, the chatbot was able to guide artificial users through the session structure and deliver the behavioral activation protocol as intended.</p><p>Weaker performance was observed when the protocol required clinical judgment and active course correction. Explaining positive reinforcement was the only component below the threshold, and helping users observe activity-mood connections was only slightly above it. Experts also noted that the chatbot often accepted proposed activities, rewards, or plans without checking their suitability, feasibility, or tailoring to the user. The 2 lowest-rated sessions showed related problems. In 1 session, the artificial user treated the chatbot as an information source; in the other, the artificial user repeatedly questioned whether the intervention would help. The chatbot responded to these turns but struggled to bring the conversation back to the behavioral activation task and complete the session as intended. These findings point to a clear refinement target: the chatbot must do more than follow the sequence. It needs to judge user input, ask useful follow-up questions, and recover when users redirect or challenge the therapeutic task.</p><p>The therapeutic capability ratings support this interpretation. Experts rated message safety, clarity, and objective and nonjudgmental communication highly, and they did not identify overtly unsafe messages in the evaluated sessions. The ratings were lower and more variable for therapeutic rapport, natural conversation flow, validation, and responsiveness to user concerns. The chatbot therefore performed best when the task was structured and less effectively when the session required clinical reasoning, relational sensitivity, or active redirection.</p><p>The evaluation approach shaped what could be learned from the data. Artificial user sessions allowed clinical experts to inspect complete chatbot-led behavioral activation sessions under controlled conditions. Therefore, the ratings identified how well the chatbot delivered the intended session in this evaluation setting and where its delivery broke down. They did not show how human users would experience the chatbot or how it would perform with more complex clinical presentations.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>Prior work has examined LLM-supported cognitive restructuring [<xref ref-type="bibr" rid="ref35">35</xref>], behavior change support [<xref ref-type="bibr" rid="ref36">36</xref>], and broader applications of LLMs in mental health care [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. Reviews suggest that LLM-based systems can generate fluent, supportive, and clinically relevant language. However, they also noted that evaluation methods remain heterogeneous, often nonstandardized, and rarely tied to established clinical quality criteria [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref38">38</xref>]. Our study extends this work by evaluating a complete LLM-supported therapeutic session using an intervention-specific fidelity scale comparable to those used to assess psychotherapists. This approach shows which parts of behavioral activation were delivered with fidelity and which parts remained vulnerable.</p><p>These findings align with prior work showing both the promise and limits of LLM-supported therapeutic tasks [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. Studies on LLM-supported cognitive restructuring and behavior-change support suggest that these systems can follow structured intervention steps, generate supportive responses, and guide users through therapeutic exercises [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. Simultaneously, reviews and clinical evaluation frameworks caution that fluent therapeutic language does not necessarily imply sound clinical judgment, reliable adaptation, or adequate handling of complex user input [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref38">38</xref>-<xref ref-type="bibr" rid="ref40">40</xref>]. Recent comparative work on explainable AI for mental health detection from social media points to a similar caution: LLM-based outputs can appear coherent and informative, but their explanations and reasoning should not be treated as self-validating evidence of clinically reliable interpretation [<xref ref-type="bibr" rid="ref41">41</xref>]. Our findings support this distinction in the context of behavioral activation. The chatbot moved through the protocol well but struggled when user input required interpretation, follow-up, or redirection.</p><p>The therapeutic capability ratings sharpen this distinction. Prior studies have examined whether users can form bonds with chatbots [<xref ref-type="bibr" rid="ref42">42</xref>] and whether chatbots can reproduce important empathic functions [<xref ref-type="bibr" rid="ref43">43</xref>]. In our study, experts rated message safety, clarity, and objective, nonjudgmental communication highly but rated therapeutic rapport, natural conversational flow, validation, and responsiveness lower and with greater variability. Therefore, clear and supportive language should not be equated with therapeutic responsiveness. A chatbot can sound safe and helpful while still missing opportunities for emotional attunement, follow-ups, or therapeutic redirection. Our findings support a more differentiated view of LLM-based therapeutic capabilities, in which procedural delivery, safe communication, relational responsiveness, and clinical judgment are evaluated separately.</p><p>This study also contributes an intervention-specific fidelity evaluation approach for LLM-based mental health chatbots. Existing chatbot evaluations often focus on single-turn prompts, general response-appropriateness metrics, usability, or symptom outcomes, with limited evidence on whether the systems deliver the intended intervention components across a full session [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. Our study extends this work by combining artificial user sessions, clinical expert assessment, and the Q-BAS to examine intervention fidelity in complete chatbot-led behavioral activation sessions. It also connects to emerging work on artificial users and LLM-simulated patients, which have used LLM-generated personas, simulated patients, and role-play interactions for controlled evaluation, counselor training, and expert-designed patient simulation [<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref21">21</xref>]. Our approach identified the average intervention fidelity, component-level weaknesses, session-level failure modes, and prompt-level refinement targets. Simulated sessions with artificial-user ratings show that plausible-looking simulated sessions do not, by themselves, establish that artificial users are realistic or sufficiently challenging replacements for human users. Experts rated the artificial users slightly below the scale midpoint for authenticity and low in difficulty, suggesting that this setup was useful for exploratory evaluation and chatbot refinement but not sufficient to represent difficult, emotionally complex, or clinically realistic human interactions.</p><p>Taken together, these findings position the chatbot as a bounded support tool rather than a replacement for therapists. This framing aligns with human-centered AI agent research, which emphasizes that health care deployment depends not only on technical performance but also on usability, trust, interpretability, ethical alignment, and fit with clinical workflows [<xref ref-type="bibr" rid="ref46">46</xref>]. Future development should focus on supervised use, with clear oversight and escalation pathways. The next step is to strengthen the weak components identified here, test the revised chatbot in more challenging artificial-user and safety-critical scenarios, and evaluate it in supervised studies with human users.</p></sec><sec id="s4-3"><title>Implications for the Design of LLM-Based Mental Health Chatbots</title><p>The expert assessment generated specific refinement targets for the chatbot. <xref ref-type="table" rid="table4">Table 4</xref> summarizes the 4 prompt-level patterns derived from the quantitative fidelity ratings and clinical experts&#x2019; qualitative feedback. These patterns identify where the current chatbot can be improved and what future iterations should be tested.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Design implications derived from expert evaluation findings.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Observed shortcoming</td><td align="left" valign="bottom">Refinement pattern</td><td align="left" valign="bottom">Illustrative example</td></tr></thead><tbody><tr><td align="left" valign="top">Explaining positive reinforcement was the only below-threshold component and showed the highest variability among all 14 components.</td><td align="left" valign="top">Granular task breakdown: convert high-level directives into sequential steps with explicit completion criteria to reduce generation variability</td><td align="left" valign="top">Replace &#x201C;explain positive reinforcement&#x201D; with a 3-step sequence: &#x201C;(1) define reinforcement with a relatable analogy, (2) contrast natural vs self-chosen rewards with concrete examples, and (3) verify understanding before proceeding&#x201D;</td></tr><tr><td align="left" valign="top">The monitoring instructions were vague and inconsistently delivered.</td><td align="left" valign="top">Template-based content: for outputs requiring a specific format, provide a ready-to-use template rather than relying on unconstrained generation</td><td align="left" valign="top">Embed a fixed tracking template: &#x201C;for each activity, note: (1) What I did, (2) When, (3) Mood before (0&#x2010;10), (4) Mood after (0&#x2010;10), and (5) What I noticed&#x201D;</td></tr><tr><td align="left" valign="top">The chatbot failed to verify whether the activities or rewards were therapeutically appropriate.</td><td align="left" valign="top">Embedded clinical decision rules: for judgment-dependent tasks, specify explicit screening criteria and conditional responses for therapeutically risky inputs</td><td align="left" valign="top">&#x201C;If the user proposes a food-based reward, validate the preference, then prompt exploration of at least 1 alternative: 'that sounds enjoyable&#x2014;let&#x2019;s also find a nonfood option so you have a backup for harder days&#x201D;</td></tr><tr><td align="left" valign="top">Two sessions were converted into unstructured FAQ<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> exchanges when users adopted information-seeking or skeptical probing stances; the chatbot responded reactively without reconnecting to the protocol and showed limited capacity for course correction once therapeutic logic was disrupted</td><td align="left" valign="top">Explicit redirection protocols: when users steer off-protocol, conditional logic that briefly acknowledges the request before reconnecting to the current therapeutic task can prevent protocol abandonment</td><td align="left" valign="top">&#x201C;If a user asks a broad self-help question midsession, acknowledge briefly: 'that&#x2019;s something the plan we&#x2019;re building is designed to help with&#x2014;let&#x2019;s keep going so we get there&#x201D;</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>FAQ: frequently asked questions.</p></fn></table-wrap-foot></table-wrap><p>These design implications can guide the iterative development of chatbots. After prompt revisions, model updates, or fixes to previously observed failures, developers can repeat the same artificial user sessions or run targeted variants of them. Expert ratings can then indicate whether the revised chatbot preserves the intended behavioral activation components, improves previously weak behaviors, or introduces new problems.</p></sec><sec id="s4-4"><title>Limitations</title><p>This study evaluated a prompt-engineered GPT-4o behavioral activation chatbot in simulated sessions as the first evaluation step before further refinement and human testing. Therefore, the findings are limited to chatbot performance under these conditions and should not be generalized to human-user performance. Because the study did not include a reference condition, the Q-BAS scores could not show how the chatbot compared with clinical experts delivering the same protocol. The Q-BAS threshold was adopted from the Q-BAS and its previous use in human-delivered behavioral activation and has not been validated for chatbot-delivered behavioral activation. The artificial users also likely behaved more cooperatively than many real users would, a known concern when LLMs are used as proxies for human participants [<xref ref-type="bibr" rid="ref47">47</xref>]. This cooperation may have inflated fidelity estimates, especially for relational and adaptive components, which would likely be more difficult with resistant, distressed, or less structured human users. Future evaluations should compare artificial user sessions with clinical expert delivery, simulations using more resistant artificial users, and supervised human user testing.</p><p>The safety findings were limited to the scenarios that occurred during testing. Experts rated the chatbot messages as highly safe, and no evaluated session contained an overtly unsafe chatbot response. However, the crisis protocol was not activated because the high-risk artificial user did not explicitly disclose suicidal ideation, intent, or self-harm thoughts during the generated session, even though the underlying personas were specified as having suicidality (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Therefore, artificial-user testing can identify some failure modes but cannot establish safety under real distress, resistance, or crisis disclosure. Future evaluations should systematically test crisis responses rather than relying on whether risks surface. This means constructing predefined high-risk scripts that require the chatbot to implement safety behavior&#x2014;for example, artificial users that escalate from ambiguous hopelessness to explicit suicidal ideation and scenarios covering self-harm intent, abuse or coercion, psychosis-related or mania-related disclosures, identity-related distress, and boundary-crossing therapeutic requests. Each script can then be scored against a prespecified expected response, such as whether the crisis referral protocol triggers at the intended threshold and at what point in the conversation. Combining such scripted edge-case tests with red-team evaluation and, subsequently, supervised studies with human users would allow crisis-response capability to be assessed directly and more robustly [<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref48">48</xref>].</p><p>Several limitations are associated with the model, implementation, and generated interactions. GPT-4o generated both the chatbot and artificial-user responses, which may have produced linguistically and behaviorally compatible interactions and reduced observable misunderstandings, ambiguity, resistance, or off-protocol behavior. Each transcript also represents a stochastic interaction drawn from the configured chatbot and artificial user prompts. We did not estimate run-to-run variance by regenerating sessions from the same personas. Future evaluations should therefore test different model simulations, with repeated generations from the same artificial users.</p><p>The rating design also limits the precision with which we can separate chatbot performance from rater differences. Each full session was rated by a clinical expert, and a subset of 18 dialogues was rated a second time to provide preliminary information on rating consistency. In this subset, the absolute agreement for the Q-BAS mean was moderate (ICC [2,1]=0.55, 95% CI 0.15-0.80). Several factors likely contributed to this result. The subset was small, which widened the CI. ICC (2,1) is a stringent absolute-agreement index that incorporates systematic differences in rating severity between raters into the reliability estimate [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>]. Licensed psychotherapists rated sessions descriptively higher than trainees, suggesting that differences in rater background may also have contributed to lower agreement. Many Q-BAS components also require subjective clinical judgment regarding whether a delivered behavior is adequate, which leaves room for legitimate disagreement. This moderate agreement means that component-level Q-BAS estimates should be interpreted as approximate and that the variance attributed to transcript-level differences may partly reflect rater severity or interpretation. More structural findings, completion of all 7 phases, and stronger ratings for clearly procedural components depend less on fine rater calibration and are correspondingly more robust. A larger multirater design is needed to estimate interrater reliability more precisely and strengthen component-level conclusions.</p><p>Finally, the findings are bounded by the study&#x2019;s LLM, language, intervention, and sample size. The evaluation used a single model (GPT-4o), German-language interactions, and a structured intervention (behavioral activation); therefore, transferring the framework to other settings should not be assumed to be straightforward. A different model or prompting strategy would require its prompts to be re-engineered and retested because the fidelity patterns and failure modes we observed were tied to this implementation. A different language would require translation and cultural adaptation of both the intervention and the artificial users, together with rechecking that safety behaviors, such as crisis referral, still trigger correctly. A different psychotherapeutic approach may not map onto a component-based fidelity scale, such as the Q-BAS, and could require a different evaluation instrument and interaction design. Behavioral activation was well-suited here precisely because its structured, protocol-driven format made complete sessions measurable with a standardized scale. The sample of 48 sessions was determined by practical constraints rather than a formal power analysis [<xref ref-type="bibr" rid="ref51">51</xref>]. Therefore, exploratory subgroup analyses in this small sample should be interpreted descriptively. Overall, the sample supported the clinical fidelity assessment and helped identify refinement targets. However, future studies should be appropriately designed to test subgroup differences and examine how well the framework transfers across models, languages, and interventions.</p></sec><sec id="s4-5"><title>Conclusion</title><p>We evaluated a GPT-4o&#x2013;powered behavioral activation chatbot across 48 artificial user sessions rated by clinical experts. The chatbot completed all intervention phases and received stronger ratings for structured procedural components than for those requiring clinical judgment, therapeutic rapport, and adaptive redirection. The findings identify concrete refinement targets before human testing, especially more systematic verification of activity plans, rewards, and user-related concerns.</p><p>Artificial user simulations, combined with expert fidelity ratings, can identify protocol-level weaknesses before human testing. In this study, this value was clearest for structured behavioral activation components, interaction breakdowns in skeptical or information-seeking sessions, and safety scenarios that required more targeted testing. The next step is a staged evaluation that tests whether the observed fidelity patterns remain stable across different simulation models, human role-play, supervised studies with human users, and targeted safety scenarios.</p><p>For developers of similar systems, the findings point to 4 practical prompt-level refinement patterns: granular task breakdown, template-based content, embedded clinical decision rules, and explicit redirection mechanisms. These patterns should be treated as refinement hypotheses for future testing, rather than validated design principles. Complete prompts for both the behavioral activation chatbot and artificial users are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> to support replication and further evaluation.</p></sec></sec></body><back><ack><p>Generative AI was used during manuscript preparation; Claude (Anthropic; Claude Opus 4.8, 5) and ChatGPT/Codex (OpenAI; GPT 5.4, 5.5, 5.6) were used for proofreading, editing, and language polishing. All AI-assisted edits were reviewed by the authors, who take full responsibility for the final manuscript.</p></ack><notes><sec><title>Funding</title><p>We acknowledge support from the KIT Publication Fund of the Karlsruhe Institute of Technology.</p></sec></notes><fn-group><fn fn-type="conflict"><p>SL received consultancy fees from companies for advice on study and intervention design in the context of e-mental health. He has also received payments for lectures on e-mental health. The other authors declare no conflicts of interest.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BH</term><def><p>Benjamini-Hochberg</p></def></def-item><def-item><term id="abb2">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb3">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb4">PHQ-9</term><def><p>Patient Health Questionnaire-9</p></def></def-item><def-item><term id="abb5">Q-BAS</term><def><p>Quality of Behavioral Activation Scale</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Torous</surname><given-names>J</given-names> </name><name name-style="western"><surname>Linardon</surname><given-names>J</given-names> </name><name name-style="western"><surname>Goldberg</surname><given-names>SB</given-names> </name><etal/></person-group><article-title>The evolving field of digital mental health: current evidence and implementation issues for smartphone apps, generative artificial intelligence, and virtual reality</article-title><source>World Psychiatry</source><year>2025</year><month>06</month><volume>24</volume><issue>2</issue><fpage>156</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.1002/wps.21299</pub-id><pub-id pub-id-type="medline">40371757</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fitzpatrick</surname><given-names>KK</given-names> </name><name name-style="western"><surname>Darcy</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vierhile</surname><given-names>M</given-names> </name></person-group><article-title>Delivering cognitive behavior therapy to young adults with symptoms of depression and anxiety using a fully automated conversational agent (Woebot): a randomized controlled trial</article-title><source>JMIR Ment Health</source><year>2017</year><month>06</month><day>6</day><volume>4</volume><issue>2</issue><fpage>e19</fpage><pub-id pub-id-type="doi">10.2196/mental.7785</pub-id><pub-id pub-id-type="medline">28588005</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Inkster</surname><given-names>B</given-names> </name><name name-style="western"><surname>Sarda</surname><given-names>S</given-names> </name><name name-style="western"><surname>Subramanian</surname><given-names>V</given-names> </name></person-group><article-title>An empathy-driven, conversational artificial intelligence agent (Wysa) for digital mental well-being: real-world data evaluation mixed-methods study</article-title><source>JMIR mHealth uHealth</source><year>2018</year><month>11</month><day>23</day><volume>6</volume><issue>11</issue><fpage>e12106</fpage><pub-id pub-id-type="doi">10.2196/12106</pub-id><pub-id pub-id-type="medline">30470676</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chan</surname><given-names>WW</given-names> </name><name name-style="western"><surname>Fitzsimmons-Craft</surname><given-names>EE</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>AC</given-names> </name><etal/></person-group><article-title>The challenges in designing a prevention chatbot for eating disorders: observational study</article-title><source>JMIR Form Res</source><year>2022</year><month>01</month><day>19</day><volume>6</volume><issue>1</issue><fpage>e28003</fpage><pub-id pub-id-type="doi">10.2196/28003</pub-id><pub-id pub-id-type="medline">35044314</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Linardon</surname><given-names>J</given-names> </name><name name-style="western"><surname>Torous</surname><given-names>J</given-names> </name><name name-style="western"><surname>Firth</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cuijpers</surname><given-names>P</given-names> </name><name name-style="western"><surname>Messer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Fuller-Tyszkiewicz</surname><given-names>M</given-names> </name></person-group><article-title>Current evidence on the efficacy of mental health smartphone apps for symptoms of depression and anxiety. a meta-analysis of 176 randomized controlled trials</article-title><source>World Psychiatry</source><year>2024</year><month>02</month><volume>23</volume><issue>1</issue><fpage>139</fpage><lpage>149</lpage><pub-id pub-id-type="doi">10.1002/wps.21183</pub-id><pub-id pub-id-type="medline">38214614</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haque</surname><given-names>MDR</given-names> </name><name name-style="western"><surname>Rubya</surname><given-names>S</given-names> </name></person-group><article-title>An overview of chatbot-based mobile mental health apps: insights from app description and user reviews</article-title><source>JMIR mHealth uHealth</source><year>2023</year><month>05</month><day>22</day><volume>11</volume><fpage>e44838</fpage><pub-id pub-id-type="doi">10.2196/44838</pub-id><pub-id pub-id-type="medline">37213181</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Blease</surname><given-names>C</given-names> </name><name name-style="western"><surname>Torous</surname><given-names>J</given-names> </name></person-group><article-title>ChatGPT and mental healthcare: balancing benefits with risks of harms</article-title><source>BMJ Ment Health</source><year>2023</year><month>11</month><volume>26</volume><issue>1</issue><fpage>e300884</fpage><pub-id pub-id-type="doi">10.1136/bmjment-2023-300884</pub-id><pub-id pub-id-type="medline">37949485</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stade</surname><given-names>EC</given-names> </name><name name-style="western"><surname>Stirman</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Ungar</surname><given-names>LH</given-names> </name><etal/></person-group><article-title>Large language models could change the future of behavioral healthcare: a proposal for responsible development and evaluation</article-title><source>Npj Ment Health Res</source><year>2024</year><month>04</month><day>2</day><volume>3</volume><issue>1</issue><fpage>12</fpage><pub-id pub-id-type="doi">10.1038/s44184-024-00056-z</pub-id><pub-id pub-id-type="medline">38609507</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meeks</surname><given-names>S</given-names> </name><name name-style="western"><surname>Van Haitsma</surname><given-names>K</given-names> </name><name name-style="western"><surname>Shryock</surname><given-names>SK</given-names> </name></person-group><article-title>Treatment fidelity evidence for BE-ACTIV - a behavioral intervention for depression in nursing homes</article-title><source>Aging Ment Health</source><year>2019</year><month>09</month><volume>23</volume><issue>9</issue><fpage>1192</fpage><lpage>1202</lpage><pub-id pub-id-type="doi">10.1080/13607863.2018.1484888</pub-id><pub-id pub-id-type="medline">30518246</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Thygesen</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Farrington</surname><given-names>J</given-names> </name><name name-style="western"><surname>Keen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name></person-group><article-title>Large language models for mental health applications: systematic review</article-title><source>JMIR Ment Health</source><year>2024</year><month>10</month><day>18</day><volume>11</volume><fpage>e57400</fpage><pub-id pub-id-type="doi">10.2196/57400</pub-id><pub-id pub-id-type="medline">39423368</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hatch</surname><given-names>SG</given-names> </name><name name-style="western"><surname>Goodman</surname><given-names>ZT</given-names> </name><name name-style="western"><surname>Vowels</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Correction: when ELIZA meets therapists: a Turing test for the heart and mind</article-title><source>PLOS Ment Health</source><year>2025</year><volume>2</volume><issue>8</issue><fpage>e0000426</fpage><pub-id pub-id-type="doi">10.1371/journal.pmen.0000426</pub-id><pub-id pub-id-type="medline">41662086</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Heinz</surname><given-names>MV</given-names> </name><name name-style="western"><surname>Mackin</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Trudeau</surname><given-names>BM</given-names> </name><etal/></person-group><article-title>Randomized trial of a generative AI chatbot for mental health treatment</article-title><source>NEJM AI</source><year>2025</year><month>03</month><day>27</day><volume>2</volume><issue>4</issue><pub-id pub-id-type="doi">10.1056/AIoa2400802</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hua</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Na</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A scoping review of large language models for generative tasks in mental health care</article-title><source>NPJ Digit Med</source><year>2025</year><month>04</month><day>30</day><volume>8</volume><issue>1</issue><fpage>230</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01611-4</pub-id><pub-id pub-id-type="medline">40307331</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thieme</surname><given-names>A</given-names> </name><name name-style="western"><surname>Belgrave</surname><given-names>D</given-names> </name><name name-style="western"><surname>Doherty</surname><given-names>G</given-names> </name></person-group><article-title>Machine learning in mental health: a systematic review of the HCI literature to support the development of effective and implementable ML systems</article-title><source>ACM Trans Comput-Hum Interact</source><year>2020</year><month>08</month><day>17</day><volume>27</volume><issue>5</issue><fpage>1</fpage><lpage>34</lpage><pub-id pub-id-type="doi">10.1145/3398069</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name></person-group><article-title>Toward retrieval-grounded evaluation for conversational large language model-based risk assessment</article-title><source>JMIR AI</source><year>2026</year><month>03</month><day>12</day><volume>5</volume><fpage>e90759</fpage><pub-id pub-id-type="doi">10.2196/90759</pub-id><pub-id pub-id-type="medline">41818631</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cuijpers</surname><given-names>P</given-names> </name><name name-style="western"><surname>Karyotaki</surname><given-names>E</given-names> </name><name name-style="western"><surname>Harrer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stikkelbroek</surname><given-names>Y</given-names> </name></person-group><article-title>Individual behavioral activation in the treatment of depression: a meta analysis</article-title><source>Psychother Res</source><year>2023</year><month>09</month><volume>33</volume><issue>7</issue><fpage>886</fpage><lpage>897</lpage><pub-id pub-id-type="doi">10.1080/10503307.2023.2197630</pub-id><pub-id pub-id-type="medline">37068380</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schuller</surname><given-names>A</given-names> </name><name name-style="western"><surname>Janssen</surname><given-names>D</given-names> </name><name name-style="western"><surname>Blumenr&#x00F6;ther</surname><given-names>J</given-names> </name><name name-style="western"><surname>Probst</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Schmidt</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>C</given-names> </name></person-group><article-title>Generating personas using LLMs and assessing their viability</article-title><source>Extended Abstracts CHI Conf Hum Factors Comput Syst</source><year>2024</year><fpage>1</fpage><lpage>7</lpage><pub-id pub-id-type="doi">10.1145/3613905.3650860</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Steenstra</surname><given-names>I</given-names> </name><name name-style="western"><surname>Nouraei</surname><given-names>F</given-names> </name><name name-style="western"><surname>Bickmore</surname><given-names>T</given-names> </name></person-group><article-title>Scaffolding empathy: training counselors with simulated patients and utterance-level performance visualizations</article-title><source>Proc 2025 CHI Conf Hum Factors Comput Syst</source><year>2025</year><fpage>1</fpage><lpage>22</lpage><pub-id pub-id-type="doi">10.1145/3706598.3714014</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Qiu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lan</surname><given-names>Z</given-names> </name></person-group><article-title>Interactive agents: simulating counselor-client psychological counseling via role-playing LLM-to-LLM interactions</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 28, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2408.15787</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Milani</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chiu</surname><given-names>JC</given-names> </name><etal/></person-group><article-title>PATIENT-&#x1D713;: using large language models to simulate patients for training mental health professionals</article-title><source>Proc 2024 Conf Empir Methods Nat Lang Process</source><year>2024</year><fpage>12772</fpage><lpage>12797</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.711</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Louie</surname><given-names>R</given-names> </name><name name-style="western"><surname>Nandi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Brunskill</surname><given-names>E</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name></person-group><article-title>Roleplay-doh: enabling domain-experts to create LLM-simulated patients via eliciting and adhering to principles</article-title><source>Proc 2024 Conf Empir Methods Nat Lang Process</source><year>2024</year><fpage>10570</fpage><lpage>10603</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.591</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schleider</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Weisz</surname><given-names>JR</given-names> </name></person-group><article-title>Little treatments, promising effects? Meta-analysis of single-session interventions for youth psychiatric problems</article-title><source>J Am Acad Child Adolesc Psychiatry</source><year>2017</year><month>02</month><volume>56</volume><issue>2</issue><fpage>107</fpage><lpage>115</lpage><pub-id pub-id-type="doi">10.1016/j.jaac.2016.11.007</pub-id><pub-id pub-id-type="medline">28117056</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schleider</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Mullarkey</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Fox</surname><given-names>KR</given-names> </name><etal/></person-group><article-title>A randomized trial of online single-session interventions for adolescent depression during COVID-19</article-title><source>Nat Hum Behav</source><year>2022</year><month>02</month><volume>6</volume><issue>2</issue><fpage>258</fpage><lpage>268</lpage><pub-id pub-id-type="doi">10.1038/s41562-021-01235-0</pub-id><pub-id pub-id-type="medline">34887544</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kuhlmeier</surname><given-names>FO</given-names> </name><name name-style="western"><surname>Bauch</surname><given-names>L</given-names> </name><name name-style="western"><surname>Gnewuch</surname><given-names>U</given-names> </name><name name-style="western"><surname>L&#x00FC;ttke</surname><given-names>S</given-names> </name></person-group><article-title>Designing chatbots to treat depression in youth: qualitative study</article-title><source>JMIR Hum Factors</source><year>2025</year><month>06</month><day>19</day><volume>12</volume><fpage>e66632</fpage><pub-id pub-id-type="doi">10.2196/66632</pub-id><pub-id pub-id-type="medline">40536944</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Kuhlmeier</surname><given-names>FO</given-names> </name><name name-style="western"><surname>Gnewuch</surname><given-names>U</given-names> </name><name name-style="western"><surname>L&#x00FC;ttke</surname><given-names>S</given-names> </name><name name-style="western"><surname>Brakemeier</surname><given-names>EL</given-names> </name><name name-style="western"><surname>M&#x00E4;dche</surname><given-names>A</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Drechsler</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gerber</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hevner</surname><given-names>A</given-names> </name></person-group><article-title>A personalized conversational agent to treat depression in youth and young adults &#x2013; a transdisciplinary design science research project</article-title><source>The Transdisciplinary Reach of Design Science Research: 17th International Conference on Design Science Research in Information Systems and Technology, DESRIST 2022, St Petersburg, FL, USA, June 1&#x2013;3, 2022, Proceedings</source><year>2022</year><publisher-name>Springer International Publishing</publisher-name><fpage>30</fpage><lpage>41</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-06516-3_3</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Franco D&#x2019;Souza</surname><given-names>R</given-names> </name><name name-style="western"><surname>Amanullah</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mathew</surname><given-names>M</given-names> </name><name name-style="western"><surname>Surapaneni</surname><given-names>KM</given-names> </name></person-group><article-title>Appraising the performance of ChatGPT in psychiatry using 100 clinical case vignettes</article-title><source>Asian J Psychiatr</source><year>2023</year><month>11</month><volume>89</volume><fpage>103770</fpage><pub-id pub-id-type="doi">10.1016/j.ajp.2023.103770</pub-id><pub-id pub-id-type="medline">37812998</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Borghouts</surname><given-names>J</given-names> </name><name name-style="western"><surname>Eikey</surname><given-names>E</given-names> </name><name name-style="western"><surname>Mark</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Barriers to and facilitators of user engagement with digital mental health interventions: systematic review</article-title><source>J Med Internet Res</source><year>2021</year><month>03</month><day>24</day><volume>23</volume><issue>3</issue><fpage>e24387</fpage><pub-id pub-id-type="doi">10.2196/24387</pub-id><pub-id pub-id-type="medline">33759801</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rice</surname><given-names>F</given-names> </name><name name-style="western"><surname>Riglin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lomax</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Adolescent and adult differences in major depression symptom profiles</article-title><source>J Affect Disord</source><year>2019</year><month>01</month><day>15</day><volume>243</volume><fpage>175</fpage><lpage>181</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2018.09.015</pub-id><pub-id pub-id-type="medline">30243197</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jardine</surname><given-names>J</given-names> </name><name name-style="western"><surname>Nadal</surname><given-names>C</given-names> </name><name name-style="western"><surname>Robinson</surname><given-names>S</given-names> </name><name name-style="western"><surname>Enrique</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hanratty</surname><given-names>M</given-names> </name><name name-style="western"><surname>Doherty</surname><given-names>G</given-names> </name></person-group><article-title>Between rhetoric and reality: real-world barriers to uptake and early engagement in digital mental health interventions</article-title><source>ACM Trans Comput-Hum Interact</source><year>2024</year><month>04</month><day>30</day><volume>31</volume><issue>2</issue><fpage>1</fpage><lpage>59</lpage><pub-id pub-id-type="doi">10.1145/3635472</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Gnewuch</surname><given-names>U</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Maedche</surname><given-names>A</given-names> </name></person-group><article-title>The effect of perceived similarity in dominance on customer self-disclosure to chatbots in conversational commerce</article-title><access-date>2026-08-20</access-date><conf-name>Proceedings of the 28th European Conference on Information Systems (ECIS)</conf-name><conf-date>Jun 15-17, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aisel.aisnet.org/ecis2020_rp/53">https://aisel.aisnet.org/ecis2020_rp/53</ext-link></comment></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dimidjian</surname><given-names>S</given-names> </name><name name-style="western"><surname>Goodman</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Sherwood</surname><given-names>NE</given-names> </name><etal/></person-group><article-title>A pragmatic randomized clinical trial of behavioral activation for depressed pregnant women</article-title><source>J Consult Clin Psychol</source><year>2017</year><month>01</month><volume>85</volume><issue>1</issue><fpage>26</fpage><lpage>36</lpage><pub-id pub-id-type="doi">10.1037/ccp0000151</pub-id><pub-id pub-id-type="medline">28045285</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rethorst</surname><given-names>CD</given-names> </name><name name-style="western"><surname>Trombello</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>PM</given-names> </name><etal/></person-group><article-title>Pilot evaluation on an adapted tele-behavioral activation to increase physical activity in persons with depression: a single-arm pilot study</article-title><source>BMC Psychol</source><year>2024</year><month>11</month><day>9</day><volume>12</volume><issue>1</issue><fpage>643</fpage><pub-id pub-id-type="doi">10.1186/s40359-024-02053-5</pub-id><pub-id pub-id-type="medline">39522018</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bunge</surname><given-names>EL</given-names> </name><name name-style="western"><surname>Desage</surname><given-names>C</given-names> </name></person-group><article-title>A framework for evaluating mental health artificial intelligence-based conversational agents</article-title><source>J technol behav sci</source><year>2025</year><month>04</month><day>26</day><volume>10</volume><issue>4</issue><fpage>731</fpage><lpage>739</lpage><pub-id pub-id-type="doi">10.1007/s41347-025-00519-w</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Mayring</surname><given-names>P</given-names> </name><name name-style="western"><surname>Fenzl</surname><given-names>T</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Baur</surname><given-names>N</given-names> </name><name name-style="western"><surname>Blasius</surname><given-names>J</given-names> </name></person-group><article-title>Qualitative inhaltsanalyse</article-title><source>Handbuch Methoden Der Empirischen Sozialforschung [Book in German]</source><year>2019</year><edition>2</edition><publisher-name>Springer Fachmedien Wiesbaden</publisher-name><fpage>633</fpage><lpage>648</lpage><pub-id pub-id-type="doi">10.1007/978-3-658-21308-4_42</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sharma</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rushton</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>IW</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Althoff</surname><given-names>T</given-names> </name></person-group><article-title>Facilitating self-guided mental health interventions through human-language model interaction: a case study of cognitive restructuring</article-title><source>Proc ACM Conf Hum Factors Comput Syst</source><year>2024</year><fpage>1</fpage><lpage>29</lpage><pub-id pub-id-type="doi">10.1145/3613904.3642761</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meyer</surname><given-names>S</given-names> </name><name name-style="western"><surname>Elsweiler</surname><given-names>D</given-names> </name></person-group><article-title>LLM-based conversational agents for behaviour change support: a randomised controlled trial examining efficacy, safety, and the role of user behaviour</article-title><source>Int J Hum Comput Stud</source><year>2025</year><month>05</month><volume>200</volume><fpage>103514</fpage><pub-id pub-id-type="doi">10.1016/j.ijhcs.2025.103514</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Campellone</surname><given-names>TR</given-names> </name><name name-style="western"><surname>Flom</surname><given-names>M</given-names> </name><name name-style="western"><surname>Montgomery</surname><given-names>RM</given-names> </name><etal/></person-group><article-title>Safety and user experience of a generative artificial intelligence digital mental health intervention: exploratory randomized controlled trial</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>23</day><volume>27</volume><fpage>e67365</fpage><pub-id pub-id-type="doi">10.2196/67365</pub-id><pub-id pub-id-type="medline">40408143</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Bhanushali</surname><given-names>T</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Badami</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hightow-Weidman</surname><given-names>L</given-names> </name></person-group><article-title>Evaluating generative AI in mental health: systematic review of capabilities and limitations</article-title><source>JMIR Ment Health</source><year>2025</year><month>05</month><day>15</day><volume>12</volume><issue>1</issue><fpage>e70014</fpage><pub-id pub-id-type="doi">10.2196/70014</pub-id><pub-id pub-id-type="medline">40373033</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qiu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Quantifying the reasoning abilities of LLMs on clinical cases</article-title><source>Nat Commun</source><year>2025</year><month>11</month><day>6</day><volume>16</volume><issue>1</issue><fpage>9799</fpage><pub-id pub-id-type="doi">10.1038/s41467-025-64769-1</pub-id><pub-id pub-id-type="medline">41198657</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grabb</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lamparth</surname><given-names>M</given-names> </name><name name-style="western"><surname>Vasan</surname><given-names>N</given-names> </name></person-group><article-title>Risks from language models for automated mental healthcare: ethics and structure for implementation (extended abstract)</article-title><source>Proc AAAI ACM Conf AI Ethics Soc</source><year>2024</year><volume>7</volume><issue>1</issue><fpage>519</fpage><pub-id pub-id-type="doi">10.1609/aies.v7i1.31654</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Xie</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>Z</given-names> </name></person-group><article-title>Explainable AI for mental health detection from social media: a comparative study of traditional machine learning and a large language model</article-title><source>SSRN</source><comment>Preprint posted online on  Mar 14, 2026</comment><pub-id pub-id-type="doi">10.2139/ssrn.6429778</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Darcy</surname><given-names>A</given-names> </name><name name-style="western"><surname>Daniels</surname><given-names>J</given-names> </name><name name-style="western"><surname>Salinger</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wicks</surname><given-names>P</given-names> </name><name name-style="western"><surname>Robinson</surname><given-names>A</given-names> </name></person-group><article-title>Evidence of human-level bonds established with a digital conversational agent: cross-sectional, retrospective observational study</article-title><source>JMIR Form Res</source><year>2021</year><month>05</month><day>11</day><volume>5</volume><issue>5</issue><fpage>e27868</fpage><pub-id pub-id-type="doi">10.2196/27868</pub-id><pub-id pub-id-type="medline">33973854</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rubin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Arnon</surname><given-names>H</given-names> </name><name name-style="western"><surname>Huppert</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Perry</surname><given-names>A</given-names> </name></person-group><article-title>Considering the role of human empathy in AI-driven therapy</article-title><source>JMIR Ment Health</source><year>2024</year><month>06</month><day>11</day><volume>11</volume><fpage>e56529</fpage><pub-id pub-id-type="doi">10.2196/56529</pub-id><pub-id pub-id-type="medline">38861302</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ding</surname><given-names>H</given-names> </name><name name-style="western"><surname>Simmich</surname><given-names>J</given-names> </name><name name-style="western"><surname>Vaezipour</surname><given-names>A</given-names> </name><name name-style="western"><surname>Andrews</surname><given-names>N</given-names> </name><name name-style="western"><surname>Russell</surname><given-names>T</given-names> </name></person-group><article-title>Evaluation framework for conversational agents with artificial intelligence in health interventions: a systematic scoping review</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>02</month><day>16</day><volume>31</volume><issue>3</issue><fpage>746</fpage><lpage>761</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad222</pub-id><pub-id pub-id-type="medline">38070173</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kocaballi</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Quiroz</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Rezazadegan</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Responses of conversational agents to health and lifestyle prompts: investigation of appropriateness and presentation structures</article-title><source>J Med Internet Res</source><year>2020</year><month>02</month><day>9</day><volume>22</volume><issue>2</issue><fpage>e15823</fpage><pub-id pub-id-type="doi">10.2196/15823</pub-id><pub-id pub-id-type="medline">32039810</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sherwood</surname><given-names>J</given-names> </name><name name-style="western"><surname>Aleisa</surname><given-names>N</given-names> </name><name name-style="western"><surname>Damoah</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Qu</surname><given-names>X</given-names> </name></person-group><article-title>Human-centered AI agents for healthcare and education: a systematic literature review</article-title><access-date>2026-06-30</access-date><conf-name>Human-Computer Interaction International (HCII)</conf-name><conf-date>Jun 22-27, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://faculty.cs.gwu.edu/xiaodongqu/papers/HCII_2025_5774_AI_Agent.pdf">https://faculty.cs.gwu.edu/xiaodongqu/papers/HCII_2025_5774_AI_Agent.pdf</ext-link></comment></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kapania</surname><given-names>S</given-names> </name><name name-style="western"><surname>Agnew</surname><given-names>W</given-names> </name><name name-style="western"><surname>Eslami</surname><given-names>M</given-names> </name><name name-style="western"><surname>Heidari</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fox</surname><given-names>SE</given-names> </name></person-group><article-title>Simulacrum of stories: examining large language models as qualitative research participants</article-title><source>Proc 2025 CHI Conf Hum Factors Comput Syst</source><year>2025</year><fpage>1</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1145/3706598.3713220</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Ganguli</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lovitt</surname><given-names>L</given-names> </name><name name-style="western"><surname>Kernion</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Red teaming language models to reduce harms: methods, scaling behaviors, and lessons learned</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 23, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2209.07858</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koo</surname><given-names>TK</given-names> </name><name name-style="western"><surname>Li</surname><given-names>MY</given-names> </name></person-group><article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research</article-title><source>J Chiropr Med</source><year>2016</year><month>06</month><volume>15</volume><issue>2</issue><fpage>155</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id><pub-id pub-id-type="medline">27330520</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McGraw</surname><given-names>KO</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>SP</given-names> </name></person-group><article-title>Forming inferences about some intraclass correlation coefficients</article-title><source>Psychol Methods</source><year>1996</year><month>03</month><volume>1</volume><issue>1</issue><fpage>30</fpage><lpage>46</lpage><pub-id pub-id-type="doi">10.1037/1082-989X.1.1.30</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lakens</surname><given-names>D</given-names> </name></person-group><article-title>Sample size justification</article-title><source>Collabra Psychol</source><year>2022</year><month>03</month><day>22</day><volume>8</volume><issue>1</issue><pub-id pub-id-type="doi">10.1525/collabra.33267</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Results, prompt-refinement hypotheses, chatbot prompt, artificial user persona, and variation expressions for the behavioral activation chatbot evaluation.</p><media xlink:href="mental_v13i1e94781_app1.pdf" xlink:title="PDF File, 640 KB"/></supplementary-material></app-group></back></article>