<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Ment Health</journal-id><journal-id journal-id-type="publisher-id">mental</journal-id><journal-id journal-id-type="index">16</journal-id><journal-title>JMIR Mental Health</journal-title><abbrev-journal-title>JMIR Ment Health</abbrev-journal-title><issn pub-type="epub">2368-7959</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v13i1e93672</article-id><article-id pub-id-type="doi">10.2196/93672</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Effectiveness of Chatbots in Mental Health Screening and Assessment: Systematic Review</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Morales Gil</surname><given-names>Isabel</given-names></name><degrees>PsyD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Marti-Estevez</surname><given-names>In&#x00E9;s</given-names></name><degrees>Prof Dr Med</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Doval</surname><given-names>Sandra</given-names></name><degrees>PsyD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Guti&#x00E9;rrez-Carrillo</surname><given-names>Jos&#x00E9; Miguel</given-names></name><degrees>Prof Dr Med</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Fausor</surname><given-names>Roc&#x00ED;o</given-names></name><degrees>PsyD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Arribas-Garc&#x00ED;a</surname><given-names>Silvia</given-names></name><degrees>PsyD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>&#x00C1;lvarez-Mon</surname><given-names>Miguel &#x00C1;ngel</given-names></name><degrees>Prof Dr Med</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="aff" rid="aff9">9</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Quintero</surname><given-names>Javier</given-names></name><degrees>Prof Dr Med</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="aff" rid="aff9">9</xref></contrib></contrib-group><aff id="aff1"><institution>Universidad Internacional De La Rioja</institution><addr-line>Avenida de la Paz</addr-line><addr-line>Logro&#x00F1;o</addr-line><addr-line>La Rioja</addr-line><country>Spain</country></aff><aff id="aff2"><institution>Hospital Infantil Universitario Ni&#x00F1;o Jes&#x00FA;s</institution><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff3"><institution>Gij&#x00F3;n Hospital</institution><addr-line>Gij&#x00F3;n</addr-line><country>Spain</country></aff><aff id="aff4"><institution>Valencian International University</institution><addr-line>Valencia</addr-line><country>Spain</country></aff><aff id="aff5"><institution>Department of Legal Medicine and Psychiatry, Complutense University</institution><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff6"><institution>Department of Psychiatry, Infanta Leonor University Hospital</institution><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff7"><institution>Ram&#x00F3;n y Cajal Institute of Sanitary Research (IRYCIS), Ram&#x00F3;n y Cajal Hospital</institution><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff8"><institution>CIBERSAM-ISCIII (Biomedical Research Networking Centre in Mental Health)</institution><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff9"><institution>Foundation for Biomedical Research and Innovation Hospital Universitario Unfanta Leonor Sureste</institution><addr-line>Madrid</addr-line><country>Spain</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Torous</surname><given-names>John</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Sabbineni</surname><given-names>Hemalatha</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Subotic-Kerry</surname><given-names>Mirjana</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Isabel Morales Gil, PsyD, Universidad Internacional De La Rioja, Avenida de la Paz, Logro&#x00F1;o, La Rioja, Spain, +34 941 20 97 43; <email>isabel.moralesgil@unir.net</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>6</day><month>10</month><year>2026</year></pub-date><volume>13</volume><elocation-id>e93672</elocation-id><history><date date-type="received"><day>26</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>27</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>28</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Isabel Morales Gil, In&#x00E9;s Marti-Estevez, Sandra Doval, Jos&#x00E9; Miguel Guti&#x00E9;rrez-Carrillo, Roc&#x00ED;o Fausor, Silvia Arribas-Garc&#x00ED;a, Miguel &#x00C1;ngel &#x00C1;lvarez-Mon, Javier Quintero. Originally published in JMIR Mental Health (<ext-link ext-link-type="uri" xlink:href="https://mental.jmir.org">https://mental.jmir.org</ext-link>), 6.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Mental Health, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mental.jmir.org/">https://mental.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mental.jmir.org/2026/1/e93672"/><abstract><sec><title>Background</title><p>Mental health disorders affect 970 million people globally; yet, over 50% do not access timely evaluation due to structural barriers and professional shortages. Chatbots and AI-based conversational agents have emerged as promising tools for mental health screening and assessment.</p></sec><sec><title>Objective</title><p>This study systematically evaluated the effectiveness, accuracy, reliability, and acceptability of chatbots and AI-based conversational agents for mental health screening and assessment in adults.</p></sec><sec sec-type="methods"><title>Methods</title><p>Systematic search conducted in May 2025 across PubMed/MEDLINE, PsycINFO, Scopus, and Web of Science (2019&#x2010;2025), following PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 guidelines. Eligible studies evaluated chatbots or AI for mental health screening/assessment in adults (&#x2265;18 y). Risk of bias was assessed using appropriate tools (Risk of Bias 2, Quality Assessment of Diagnostic Accuracy Studies-2, Joanna Briggs Institute checklists, and Mixed Methods Appraisal Tool). This systematic review was registered with PROSPERO (International Prospective Register of Systematic Reviews; CRD420251072392).</p></sec><sec sec-type="results"><title>Results</title><p>Eighteen studies (2021&#x2010;2025) were included, with samples ranging from 20 to 3902 participants. Rule-based chatbots demonstrated high reliability (Cronbach &#x03B1; &#x003E;0.85) and good acceptability (Acceptability of Intervention Measure &#x003E;19/25). Generative models (large language models) achieved sensitivities of 0.84 to 0.93 and specificities of 0.80 to 0.96 for depression and anxiety, with correlations up to <italic>r</italic>=0.96 with expert clinicians in suicide risk assessment. Hybrid approaches combining large language models with machine learning achieved exceptional performance for cognitive impairment (<italic>F</italic><sub>1</sub>-score of 92.1%, specificity of 99.6%). Most studies reported high user satisfaction (&#x2265;70%), although barriers existed among older populations. Methodological quality was heterogeneous with a moderate risk of bias in critical dimensions.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Chatbots and AI conversational agents demonstrate clinically relevant performance in mental health screening and assessment. However, safe implementation requires clear clinical protocols, professional supervision, integration with electronic health records, and active mitigation of algorithmic bias. These technologies should complement rather than replace clinical judgment.</p></sec><sec><title>Trial Registration</title><p>PROSPERO CRD420251072392; https://www.crd.york.ac.uk/PROSPERO/view/CRD420251072392</p></sec></abstract><kwd-group><kwd>chatbots</kwd><kwd>AI</kwd><kwd>large language models</kwd><kwd>mental health</kwd><kwd>psychological assessment</kwd><kwd>screening</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Mental health disorders constitute one of the leading causes of disability worldwide, affecting 970 million people in 2019, with depressive and anxiety disorders being the most prevalent diagnoses [<xref ref-type="bibr" rid="ref1">1</xref>]. Globally, these conditions account for approximately 14% of disability-adjusted life years, with depression alone ranking as the third leading cause of disease burden [<xref ref-type="bibr" rid="ref2">2</xref>]. In Spain, it is estimated that 1 in 4 adults will experience a mental health disorder throughout their lifetime, reflecting a significant burden on the health care system [<xref ref-type="bibr" rid="ref3">3</xref>]. However, more than 50% of individuals with symptoms related to these disorders do not access timely evaluation due to structural barriers, social stigma, and a shortage of professionals [<xref ref-type="bibr" rid="ref4">4</xref>]. Average waiting times for initial psychiatric assessment can exceed 60 days in many European health systems, with significant geographical disparities between urban and rural areas [<xref ref-type="bibr" rid="ref5">5</xref>]. This treatment gap is particularly pronounced in low- and middle-income countries, where up to 85% of individuals with mental health needs receive no care at all [<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>This care gap, exacerbated by prolonged waiting lists and geographical inequalities, has driven the search for scalable digital solutions that can expand coverage and alleviate clinical care services. In this context, chatbots and conversational agents based on AI emerge as promising tools for psychological screening and assessment, offering continuous availability (24/7), replicating validated protocols, and automatically referring urgent cases to professionals [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Early systematic reviews demonstrated that conversational agents show promising efficacy and acceptability in mental health contexts, though most evidence focused on therapeutic interventions rather than assessment capabilities [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. These systems can be broadly categorized into rule-based chatbots, which follow predetermined decision trees and administer standardized psychometric scales, and generative models based on large language models (LLMs) such as ChatGPT-4o (OpenAI), Claude 3.5 (Anthropic), and Gemini 1.5 (Google), which can engage in more flexible, context-aware conversations and interpret unstructured clinical narratives [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. Rule-based systems, which dominated mental health chatbot research until 2023, offer advantages in transparency, clinical control, and consistent administration of validated instruments, with proven reliability for structured screening protocols [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. However, they lack the flexibility to adapt to individual patient narratives or handle complex, nuanced clinical presentations [<xref ref-type="bibr" rid="ref14">14</xref>]. Furthermore, the integration of prioritization algorithms and real-time data analysis enables the generation of more accurate and personalized assessments, aligning with the European Commission&#x2019;s recommendations for the digitalization of mental health care [<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>The landscape shifted dramatically with the emergence of advanced LLMs from 2023 onward. Recent systematic analyses show that LLM-based chatbots surged from 16% of studies in 2022 to 45% in 2024, becoming the most frequently studied architecture in mental health AI research [<xref ref-type="bibr" rid="ref15">15</xref>]. These models demonstrate capabilities that extend beyond simple symptom checklists: they can engage in open-ended clinical dialogue, interpret contextual information, generate psychodynamic formulations, and perform diagnostic reasoning that approaches the performance of experienced clinicians [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. For instance, GPT-4 has passed psychiatric licensing examinations and demonstrated diagnostic accuracy comparable to psychiatrists in standardized clinical scenarios [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Recent evaluations of state-of-the-art models, including DeepSeek-R1 (DeepSeek), GPT-4.1 (OpenAI), and Llama 4 (Meta Platforms, Inc), have shown exceptional performance in both mental health knowledge assessment and illness diagnosis tasks [<xref ref-type="bibr" rid="ref18">18</xref>].</p><p>In recent years, AI has demonstrated a disruptive role in mental health, with applications ranging from early symptom detection to the automation of therapeutic interventions. Recent reviews indicate that conversational systems can accurately identify linguistic and behavioral patterns associated with depression and anxiety in screening contexts, with high precision rates [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. Beyond depression and anxiety, emerging evidence suggests potential applications in detecting cognitive impairment [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref20">20</xref>], assessing suicide risk [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>], evaluating thought disorder in schizophrenia [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>], and screening for postpartum posttraumatic stress disorder (PTSD) [<xref ref-type="bibr" rid="ref25">25</xref>]. A detailed scoping review of 95 studies found that 71% of LLM applications in mental health focused on screening and detection tasks, with reported accuracies ranging from 0.80 to 0.96 for depression and anxiety classification [<xref ref-type="bibr" rid="ref7">7</xref>]. In suicide risk assessment, LLMs have demonstrated correlations with expert clinicians exceeding <italic>r</italic>=0.90 in standardized vignette studies; however, concerns persist about the potential underestimation of risk in complex cases [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]. These technologies not only expand access but also facilitate personalized, scalable, and cost-effective interventions, which are critical given the growing demand and shortage of human resources.</p><p>However, their implementation raises ethical and methodological challenges: the need for clinical supervision, protection of sensitive data, algorithmic transparency, and the absence of longitudinal studies evaluating their sustained impact [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. Of particular concern is algorithmic bias, which can arise across the AI life cycle and threaten equity in AI-assisted mental health assessment [<xref ref-type="bibr" rid="ref28">28</xref>]. Systematic investigations have revealed that AI tools for mental health screening demonstrate differential performance across demographic groups, with sensed-behavioral patterns showing inconsistent relationships with depression symptoms across age, race, and socioeconomic subgroups [<xref ref-type="bibr" rid="ref30">30</xref>]. Qualitative analyses of leading LLMs found that race-explicit or race-implied patient information frequently resulted in inferior treatment recommendations, though diagnostic decisions showed less bias [<xref ref-type="bibr" rid="ref7">7</xref>]. Natural language processing models trained on public datasets exhibit measurable biases related to religion, race, gender, nationality, sexuality, and age in their mental health terminology [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. These biases can perpetuate health disparities and undermine the equity of AI-powered mental health tools, particularly affecting marginalized populations who already experience barriers to quality care [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. Moreover, current evidence is heterogeneous and limited, focused primarily on acceptability and user experience, while key aspects such as diagnostic validity, reliability, clinical utility, and equity in access have been less explored.</p><p>Previous systematic reviews have examined AI applications in mental health interventions and user acceptability [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref19">19</xref>], but these studies largely predate the emergence of advanced generative models and do not specifically focus on assessment capabilities. Despite growing interest in the use of chatbots and AI systems for psychological assessment, the available evidence lacks a critical synthesis integrating performance metrics, acceptability, and clinical applicability in the context of recent advances in generative models. This systematic review seeks to fill that gap by evaluating the effectiveness, accuracy, reliability, and acceptability of these tools, with the aim of guiding their safe and evidence-based implementation in health care settings.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Protocol Registration</title><p>This systematic review was registered in PROSPERO (International Prospective Register of Systematic Reviews) on August 2, 2025 (registration number CRD420251072392). The review followed the PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 guidelines and was structured according to the PIO (Population, Intervention, Outcome) framework; the completed PRISMA 2020 checklist is provided in <xref ref-type="supplementary-material" rid="app3">Checklist 1</xref>.</p><p>.</p><p><xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref></p></sec><sec id="s2-2"><title>Eligibility Criteria</title><p>Studies were selected according to predefined criteria aligned with the research question and the PIO structure:</p><sec id="s2-2-1"><title>Inclusion Criteria</title><p>Inclusion criteria were as follows:</p><list list-type="bullet"><list-item><p>Publications between 2019 and 2025 in English or Spanish.</p></list-item><list-item><p>Adults (&#x2265;18 y) with a diagnosed or suspected psychological or psychiatric disorder.</p></list-item><list-item><p>Evaluation of chatbot or AI use for psychological screening or assessment.</p></list-item><list-item><p>Original peer-reviewed studies: randomized controlled trials (RCTs), validation studies, diagnostic accuracy studies, observational studies, qualitative studies, mixed methods studies, and pilot and feasibility studies.</p></list-item></list></sec><sec id="s2-2-2"><title>Exclusion Criteria</title><p>Exclusion criteria were as follows:</p><list list-type="bullet"><list-item><p>Studies focusing exclusively on therapeutic intervention.</p></list-item><list-item><p>General medical care without a psychological assessment component.</p></list-item><list-item><p>Conference abstracts without full text, editorials, opinion pieces.</p></list-item><list-item><p>Narrative reviews (systematic reviews included only for context).</p></list-item><list-item><p>Pediatric or adolescent population (&#x003C;18 y).</p></list-item></list></sec></sec><sec id="s2-3"><title>Information Sources and Search Strategy</title><p>The bibliographic search was conducted in May 2025 in the following electronic databases: PubMed/MEDLINE, PsycINFO, Scopus, and Web of Science Core Collection. These sources were selected for their relevance to biomedical, psychological, and multidisciplinary fields, ensuring comprehensive coverage of studies on AI and mental health.</p><p>For each database, specific search strategies were designed, combining controlled vocabulary terms (MeSH in PubMed and the Thesaurus in PsycINFO) and free-text keywords, adapted to the syntax and Boolean operators of each platform.</p><list list-type="bullet"><list-item><p>Population: adults with mental disorders or psychological symptoms;</p></list-item><list-item><p>Intervention: chatbots, conversational agents, AI applied to psychological assessment;</p></list-item><list-item><p>Outcomes: effectiveness, diagnostic accuracy, reliability, validity, acceptability, and feasibility.</p></list-item></list><p>The following limits were applied: publications between 2019 and 2025, in English or Spanish. No study design restrictions were applied in the initial search. Additionally, the reference lists of included studies and relevant systematic reviews were manually reviewed to identify additional literature (<xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Characteristics and results of included studies.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Authors (y)</td><td align="left" valign="bottom">Sample/data (N)</td><td align="left" valign="bottom">Chatbot/AI</td><td align="left" valign="bottom">Condition assessed</td><td align="left" valign="bottom">Design and<break/>reference instrument(s)</td><td align="left" valign="bottom">Key findings</td></tr></thead><tbody><tr><td align="left" valign="top">Anmella et al [<xref ref-type="bibr" rid="ref34">34</xref>], 2023</td><td align="left" valign="top">34 (primary care and health care workers)</td><td align="left" valign="top">Vickybot</td><td align="left" valign="top">Anxiety, depression, work-related burnout, and suicidal risk</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Multiphase development plus feasibility and potential effectiveness study: single-arm longitudinal study (1 mo follow-up) with self-assessments at baseline, 2 weeks, and 4 weeks</p></list-item><list-item><p>PHQ-9<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>; GAD-7<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>; MBI<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></p></list-item></list></td><td align="left" valign="top">After 2 weeks of use, no significant differences for anxiety or depressive symptoms; burnout moderately reduced (<italic>z</italic>=&#x2212;2.07, <italic>P</italic>=.04), 9% activated suicide alert</td></tr><tr><td align="left" valign="top">Bartal et al [<xref ref-type="bibr" rid="ref25">25</xref>], 2024</td><td align="left" valign="top">1295 (postpartum)</td><td align="left" valign="top">ChatGPT-3.5-turbo-16k+embeddings ADA<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup>+DNN<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top">Postpartum PTSD<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Cross-sectional research</p></list-item><list-item><p>Maternal narratives and PCL-5<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></p></list-item></list></td><td align="left" valign="top">ChatGPT-3.5&#x2010;16k zero-shots AUC<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup>=0.60, <italic>F</italic><sub>1</sub>-score=0.33, sensitivity=0.20, specificity=0.99;/few-shots AUC=0.60, <italic>F</italic><sub>1</sub>-score=0.38, sensitivity=0.24, specificity=0.96; model embeddings AUC=0.80, <italic>F</italic><sub>1</sub>-score=0.81, sensitivity=0.85, specificity=0.75</td></tr><tr><td align="left" valign="top">de Arriba-P&#x00E9;rez et al [<xref ref-type="bibr" rid="ref20">20</xref>], 2024</td><td align="left" valign="top">30 (57% with cognitive impairment)</td><td align="left" valign="top">Celia (chatbot+ML<sup><xref ref-type="table-fn" rid="table1fn9">i</xref></sup> with feature extraction using GPT LLM<sup><xref ref-type="table-fn" rid="table1fn10">j</xref></sup>)</td><td align="left" valign="top">Cognitive impairment</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Cross-sectional research</p></list-item><list-item><p>Clinical diagnosis</p></list-item></list></td><td align="left" valign="top">LLM-features+RF<sup><xref ref-type="table-fn" rid="table1fn11">k</xref></sup>: accuracy=98.47%, sensitivity (impairment)=97.78%; &#x003E;n grams (76.67%) and &#x003E;direct ChatGPT (57%&#x2010;61%)</td></tr><tr><td align="left" valign="top">Dosovitsky et al [<xref ref-type="bibr" rid="ref35">35</xref>], 2021</td><td align="left" valign="top">3895 adults over 65 years</td><td align="left" valign="top">Tess (chatbot)</td><td align="left" valign="top">Depression</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Cross-sectional research</p></list-item><list-item><p>A chatbot version of PHQ-9</p></list-item></list></td><td align="left" valign="top">&#x03B1;=0.896</td></tr><tr><td align="left" valign="top">Du et al [<xref ref-type="bibr" rid="ref12">12</xref>], 2024</td><td align="left" valign="top">Clinical EHRs<sup><xref ref-type="table-fn" rid="table1fn12">l</xref></sup>: 4949 sections for baseline models and 1996 sections for final testing</td><td align="left" valign="top">Llama 2 vs GPT-4+DNN+XGBoost<sup><xref ref-type="table-fn" rid="table1fn13">m</xref></sup> ensemble</td><td align="left" valign="top">Cognitive impairment</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Cross-sectional comparative study</p></list-item><list-item><p>Clinical diagnosis in EHR</p></list-item></list></td><td align="left" valign="top">Ensemble models are better than individuals. Ensemble <italic>F</italic><sub>1</sub>-score=92.2%; recall=94.2%; precision=90.3%; E=99.6%</td></tr><tr><td align="left" valign="top">He et al [<xref ref-type="bibr" rid="ref36">36</xref>], 2024</td><td align="left" valign="top">100 web-based medical consultation samples, with 239 randomly selected consultation questions</td><td align="left" valign="top">ChatGPT-4 vs ERNIE bot 2.2.3</td><td align="left" valign="top">Autism</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Cross-sectional research</p></list-item><list-item><p>Comparative assessment (physicians vs 2 AI chatbots); Clinical diagnosis</p></list-item></list></td><td align="left" valign="top">Physicians scored higher in relevance (H=111.67, <italic>P</italic>&#x003C;.001; MD<sup><xref ref-type="table-fn" rid="table1fn14">n</xref></sup>=3.75, 95% CI 3.63&#x2010;3.82) and usefulness (H=135.81, <italic>P</italic>&#x003C;.001; MD=3.54, 95% CI 3.47&#x2010;3.62); ChatGPT-4 scored higher in empathy (H=118.58, <italic>P</italic>&#x003C;.001; MD=3.64, 95% CI 3.57&#x2010;3.71); no differences between physicians and ChatGPT 4 for correctness (H=49.99, <italic>P</italic>&#x003C;.001); ERNIE obtained the lowest scores across all 4 dimensions</td></tr><tr><td align="left" valign="top">Hur et al [<xref ref-type="bibr" rid="ref37">37</xref>], 2024</td><td align="left" valign="top">467</td><td align="left" valign="top">ChatGPT-3.5/4; Linguistic Inquiry and Word Count (LIWC-22)</td><td align="left" valign="top">Depression (prognosis)</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Longitudinal (3-wk follow-up)</p></list-item><list-item><p>Open-ended questions; PHQ-9</p></list-item></list></td><td align="left" valign="top">Human-rated sentiment score predicted depression trajectory in individuals with minimal symptoms (&#x03B2;=&#x2212;0.15<sup><xref ref-type="table-fn" rid="table1fn15">o</xref></sup>, SE=0.05, <italic>t</italic>=&#x2212;2.95<sup><xref ref-type="table-fn" rid="table1fn16">p</xref></sup>, <italic>P</italic>=.003) or with mild-to-moderate symptoms (&#x03B2;=&#x2212;0.29, SE=0.09, <italic>t</italic>=&#x2212;3.23, <italic>P</italic>=.001). ChatGPT also predicted future depressive symptoms in a similar way to humans; LIWC did not predict prognosis</td></tr><tr><td align="left" valign="top">Kang and Hong [<xref ref-type="bibr" rid="ref38">38</xref>], 2025</td><td align="left" valign="top">20</td><td align="left" valign="top">HoMemeTown Dr CareSam vs Woebot/Happify</td><td align="left" valign="top">Primary (suicidal ideation and severe depression) and secondary (sleep disturbances and social withdrawal) risk indicators+seamless support</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Pilot testing of the chatbot</p></list-item><list-item><p>Cross-sectional research</p></list-item><list-item><p>One overall satisfaction question and 7 quantitative items assessing effective psychological counseling</p></list-item></list></td><td align="left" valign="top">High satisfaction: support (mean 9.0, SD 1.2); perceived empathy (mean 8.7, SD 1.6); and active listening (mean 8.0, SD 1.8). Satisfaction level higher vs Woebot and Happify (<italic>F</italic>=12.94, <italic>P</italic>&#x003C;.001)</td></tr><tr><td align="left" valign="top">Kaywan et al [<xref ref-type="bibr" rid="ref39">39</xref>], 2023</td><td align="left" valign="top">50</td><td align="left" valign="top">DEPRA (depression analysis chatbot, based on the Structured Interview Guide for the HDRS<sup><xref ref-type="table-fn" rid="table1fn17">q</xref></sup> [SIGH_D])</td><td align="left" valign="top">Depression (early detection)</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Nonclinical trial</p></list-item><list-item><p>IDS (Inventory of Depressive Symptomatology); QIDS (Quick Inventory of Depressive Symptomatology)+user satisfaction</p></list-item></list></td><td align="left" valign="top">Overall satisfaction: 79% scored 3.95/5<break/>both scoring returned to a similar outcome with slight variations for depression level</td></tr><tr><td align="left" valign="top">Kosyluk et al [<xref ref-type="bibr" rid="ref40">40</xref>], 2024</td><td align="left" valign="top">Of the 329 participants, only 222 (67.5%) used the chatbot</td><td align="left" valign="top">Tabatha chatbot</td><td align="left" valign="top">Psychological distress</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Pilot testing of the chatbot</p></list-item><list-item><p>Cross-sectional research</p></list-item><list-item><p>PHQ-9</p></list-item></list></td><td align="left" valign="top">Of the 222 individuals who used the chatbot, 168 (75.7%) completed the PHQ-9 screening and 164 (73.9%) completed the acceptability</td></tr><tr><td align="left" valign="top">Lho et al [<xref ref-type="bibr" rid="ref41">41</xref>], 2025</td><td align="left" valign="top">1064</td><td align="left" valign="top">General-purpose LLMs (GPT-4o, GPT-3.5-turbo-16k, and Gemini 1.0 Pro) and text-embedding models (OpenAI text-embedding-3-large, -3-small, and ada-002)</td><td align="left" valign="top">Clinically significant depression and high suicide risk</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Cross-sectional, retrospective observational study using SCT<sup><xref ref-type="table-fn" rid="table1fn18">r</xref></sup> narratives and self-report scales</p></list-item><list-item><p>K-BDI-II<sup><xref ref-type="table-fn" rid="table1fn19">s</xref></sup>, Zung Self-Rating Depression Scale, Beck Scale for Suicidal Ideation, and K-WAIS-IV<sup><xref ref-type="table-fn" rid="table1fn20">t</xref></sup> (FSIQ<sup><xref ref-type="table-fn" rid="table1fn21">u</xref></sup>)</p></list-item></list></td><td align="left" valign="top">LLMs and embedding-based models achieved AUROC<sup><xref ref-type="table-fn" rid="table1fn22">v</xref></sup>&#x003E;0.7 for detecting depression and suicide risk, with best performance for self-concept narratives and text-embedding-3-large+XGB (AUROC 0.84 for depression)</td></tr><tr><td align="left" valign="top">Li et al [<xref ref-type="bibr" rid="ref8">8</xref>], 2024</td><td align="left" valign="top">20</td><td align="left" valign="top">GPT-4</td><td align="left" valign="top">Adult-acquired buried penis</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Qualitative description study (semistructured interviews)</p></list-item><list-item><p>None (qualitative analysis of transcripts)</p></list-item></list></td><td align="left" valign="top">GPT-4 identified key themes (urinary/sexual/mental health issues) with moderate agreement to humans (&#x03BA;=0.401); humans found richer subthemes; AI consistent across iterations</td></tr><tr><td align="left" valign="top">Liu et al [<xref ref-type="bibr" rid="ref42">42</xref>], 2025</td><td align="left" valign="top">200</td><td align="left" valign="top">ChatGPT-4 (generating GPT-PHQ-9 and GPT-GAD-7)</td><td align="left" valign="top">Anxiety and depression</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Cross-sectional study</p></list-item><list-item><p>PHQ-9 and GAD-7</p></list-item></list></td><td align="left" valign="top">GPT-PHQ-9 and GPT-GAD-7 demonstrated good agreement (ICC<sup><xref ref-type="table-fn" rid="table1fn23">w</xref></sup> 0.80/0.70, &#x03C1; 0.63/0.68, AUC 0.96/0.86) with optimal cutoffs of 9.5 and 6.5, respectively</td></tr><tr><td align="left" valign="top">McBain et al [<xref ref-type="bibr" rid="ref21">21</xref>], 2025</td><td align="left" valign="top">N/A<sup><xref ref-type="table-fn" rid="table1fn24">x</xref></sup> (24 hypothetical scenarios; no human participants)</td><td align="left" valign="top">ChatGPT-4o, Claude 3.5 Sonnet, and Gemini 1.5 Pro</td><td align="left" valign="top">Suicidal ideation</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Observational cross-sectional study</p></list-item><list-item><p>SIRI-2 (Suicidal Ideation Response Inventory&#x2013;revised)</p></list-item></list></td><td align="left" valign="top">LLMs showed upward bias rating responses as more appropriate vs expert suicidologists (MD 0.61&#x2010;0.86); SIRI-2 scores: ChatGPT 45.7 (&#x2248;master&#x2019;s counselors), Claude 36.7 (&#x003E;trained mental health professionals), and Gemini 54.5 (&#x2248;untrained school staff)</td></tr><tr><td align="left" valign="top">Ohse et al [<xref ref-type="bibr" rid="ref43">43</xref>], 2024</td><td align="left" valign="top">51</td><td align="left" valign="top">GPT-4 (LLM model, zero-shot, and no fine-tuning)</td><td align="left" valign="top">Social anxiety (Social Anxiety Disorder [SAD])</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Cross-sectional, semistructured interviews based on LSAS<sup><xref ref-type="table-fn" rid="table1fn25">y</xref></sup> with transcript analysis.</p></list-item><list-item><p>Social Phobia Inventory (SPIN, German version), cut-off 25</p></list-item></list></td><td align="left" valign="top">High correlation (<italic>r</italic>=0.79) between GPT-4 predictions and actual SPIN; <italic>F</italic><sub>1</sub>-score accuracy=0.84 (threshold 25); AUC=0.93; optimal threshold 18</td></tr><tr><td align="left" valign="top">Shin et al [<xref ref-type="bibr" rid="ref44">44</xref>], 2024</td><td align="left" valign="top">91</td><td align="left" valign="top">GPT-3.5 and GPT-4 (con/sin fine-tuning and chain-of-thought prompting)</td><td align="left" valign="top">Depression</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Retrospective analysis of 2-week diary writing via app (EMA<sup><xref ref-type="table-fn" rid="table1fn26">z</xref></sup> text data)</p></list-item><list-item><p>PHQ-9, cut-off &#x2265;10</p></list-item><list-item><p>Beck Scale for Suicide Ideation (cut-off &#x2265;8); clinical review by psychiatrists</p></list-item></list></td><td align="left" valign="top">GPT-3.5 fine-tuning: accuracy 0.902, specificity 0.955; GPT-3.5 without fine-tuning: balanced accuracy 0.844, recall 0.929; useful diary entries for depression screening</td></tr><tr><td align="left" valign="top">Shinan-Altman et al [<xref ref-type="bibr" rid="ref22">22</xref>], 2024</td><td align="left" valign="top">No human participants; 160 AI evaluations (10 per vignette across 8 vignettes for each of ChatGPT-3.5 and ChatGPT-4)</td><td align="left" valign="top">ChatGPT-3.5 and ChatGPT-4</td><td align="left" valign="top">Suicide risk (suicidal thoughts, attempts, serious attempts, and mortality), influenced by history of depression and access to weapons</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Experimental vignette study with multivariate 3-way ANOVA (factors: depression history, weapon access, and gender) and Bonferroni post hoc tests</p></list-item><list-item><p>Custom Likert-scale questions (0&#x2010;7: very slight to very high likelihood) adapted from Levi-Belz and Gamliel [<xref ref-type="bibr" rid="ref45">45</xref>]</p></list-item></list></td><td align="left" valign="top">Both models recognized depression history as a risk factor; ChatGPT-4 showed nuanced integration of weapon access and interaction effects, assigning higher severity ratings overall vs ChatGPT-3.5.</td></tr><tr><td align="left" valign="top">So et al [<xref ref-type="bibr" rid="ref46">46</xref>], 2024</td><td align="left" valign="top">10</td><td align="left" valign="top">GPT-4 Turbo and GPT-3.5 Turbo (zero-shot, few-shot, fine-tuning, and RAG<sup><xref ref-type="table-fn" rid="table1fn27">aa</xref></sup>)</td><td align="left" valign="top">PTSD symptoms, depressive/anxiety disorders, and alcohol use disorder</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Pilot study: LLM alignment on labeled transcripts (training/validation/test split)</p></list-item><list-item><p>Metrics include recall midtoken distance, <italic>F</italic><sub>1</sub>-score, G-Eval, and BERT<sup><xref ref-type="table-fn" rid="table1fn28">ab</xref></sup> Score.</p></list-item></list></td><td align="left" valign="top">LLMs achieved &#x003E;0.8 accuracy/<italic>F</italic><sub>1</sub>-score for symptom naming (fine-tuning best); 70% segments with <italic>d</italic>&#x2264;20 for section delineation; high G-Eval (&#x003E;4.6 coherence) for summaries using stressors+symptoms</td></tr><tr><td align="left" valign="top">Voppel et al [<xref ref-type="bibr" rid="ref23">23</xref>], 2021</td><td align="left" valign="top">100 (50 Schizophrenia-spectrum disorders patients+50 controls)</td><td align="left" valign="top">word2vec semantic model</td><td align="left" valign="top">Schizophrenia-spectrum disorders</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Semistructured interview; word2vec analysis; and RF classification</p></list-item><list-item><p>PANSS<sup><xref ref-type="table-fn" rid="table1fn29">ac</xref></sup>; CASH<sup><xref ref-type="table-fn" rid="table1fn30">ad</xref></sup>/MINI<sup><xref ref-type="table-fn" rid="table1fn31">ae</xref></sup></p></list-item></list></td><td align="left" valign="top">85% classification accuracy (86% sensitivity and 84% specificity)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>PHQ-9: Patient Health Questionnaire-9.</p></fn><fn id="table1fn2"><p><sup>b</sup>GAD-7: Generalized Anxiety Disorder-7.</p></fn><fn id="table1fn3"><p><sup>c</sup>MBI: Maslach Burnout Inventory.</p></fn><fn id="table1fn4"><p><sup>d</sup>ADA: assessment of diagnostic accuracy.</p></fn><fn id="table1fn5"><p><sup>e</sup>DNN: deep neural network.</p></fn><fn id="table1fn6"><p><sup>f</sup>PTSD: posttraumatic stress disorder.</p></fn><fn id="table1fn7"><p><sup>g</sup>PCL-5: PTSD Checklist for<italic> DSM-5</italic>.</p></fn><fn id="table1fn8"><p><sup>h</sup>AUC: area under the curve.</p></fn><fn id="table1fn9"><p><sup>i</sup>ML: machine learning.</p></fn><fn id="table1fn10"><p><sup>j</sup>LLM: large language model.</p></fn><fn id="table1fn11"><p><sup>k</sup>RF: random forest.</p></fn><fn id="table1fn12"><p><sup>l</sup>EHR: electronic health record.</p></fn><fn id="table1fn13"><p><sup>m</sup>XGBoost: Extreme Gradient Boosting.</p></fn><fn id="table1fn14"><p><sup>n</sup>MD: mean difference.</p></fn><fn id="table1fn15"><p><sup>o</sup>Beta (&#x03B2;) coefficient from robust linear regression.</p></fn><fn id="table1fn16"><p><sup>p</sup><italic>t</italic> statistic associated with the estimated regression coefficient in the robust linear regression model.</p></fn><fn id="table1fn17"><p><sup>q</sup>HDRS: Hamilton Depression Rating Scale.</p></fn><fn id="table1fn18"><p><sup>r</sup>SCT: sentence completion test.</p></fn><fn id="table1fn19"><p><sup>s</sup>K-BDI-II: Korean Beck Depression Inventory&#x2013;Second Edition.</p></fn><fn id="table1fn20"><p><sup>t</sup>K-WAIS-IV: Korean Wechsler Adult Intelligence Scale&#x2013;Fourth Edition.</p></fn><fn id="table1fn21"><p><sup>u</sup>FSIQ: Full-Scale IQ.</p></fn><fn id="table1fn22"><p><sup>v</sup>AUROC: area under the receiver operating characteristic curve.</p></fn><fn id="table1fn23"><p><sup>w</sup>ICC: intraclass correlation coefficient.</p></fn><fn id="table1fn24"><p><sup>x</sup>N/A: not applicable.</p></fn><fn id="table1fn25"><p><sup>y</sup>LSAS: Liebowitz Social Anxiety Scale.</p></fn><fn id="table1fn26"><p><sup>z</sup>EMA: ecological momentary assessment.</p></fn><fn id="table1fn27"><p><sup>aa</sup>RAG: retrieval-augmented generation.</p></fn><fn id="table1fn28"><p><sup>ab</sup>BERT: Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table1fn29"><p><sup>ac</sup>PANSS: Positive and Negative Syndrome Scale.</p></fn><fn id="table1fn30"><p><sup>ad</sup>CASH: Comprehensive Assessment of Symptoms and History.</p></fn><fn id="table1fn31"><p><sup>ae</sup>MINI: Mini-International Neuropsychiatric Interview.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-4"><title>Study Selection</title><p>The selection process was carried out in two phases: (1) initial screening of titles and abstracts and (2) full-text evaluation to determine final eligibility.</p><p>Three independent reviewers conducted the screening in parallel (JMG-C, IM-E, and IMG), using the bibliographic management tool Rayyan. Discrepancies were resolved through discussion and, when necessary, with the intervention of a fourth reviewer to reach consensus.</p><p>The previously defined inclusion and exclusion criteria (<italic>Eligibility Criteria</italic>) were applied. The complete process was documented following the PRISMA 2020 guideline and is presented in the <italic>Results</italic> section (PRISMA diagram), which details the number of identified records, those eliminated due to duplication, those excluded after title/abstract review, and the reasons for exclusion in the full-text phase.</p></sec><sec id="s2-5"><title>Data Extraction and Quality Assessment</title><p>Data extraction was performed independently by 2 reviewers, using a standardized template designed for this review. For each included study, the following variables were collected:</p><list list-type="bullet"><list-item><p>General information: author, year of publication, country, and language.</p></list-item><list-item><p>Population characteristics: sample size, inclusion criteria, and diagnosis or condition evaluated.</p></list-item><list-item><p>Intervention: type of chatbot or AI system (model, platform, and architecture), interaction modality (text and voice), and context of use (clinical, community, and experimental).</p></list-item><list-item><p>Reference instruments: psychometric scales, diagnostic interviews, or a gold standard used for comparison.</p></list-item><list-item><p>Main outcomes: performance metrics (sensitivity, specificity, area under the curve [AUC], <italic>F</italic><sub>1</sub>-score, and intraclass correlation coefficient), validity and reliability indicators, acceptability, usability, and feasibility.</p><list list-type="bullet"><list-item><p>In case of discrepancies between reviewers, these were resolved by consensus or with the intervention of a third evaluator.</p></list-item></list></list-item><list-item><p>Risk of bias assessment.</p><list list-type="bullet"><list-item><p>The risk of bias was assessed using specific tools according to the methodological design of each study:</p></list-item></list></list-item><list-item><p>RCTs: Cochrane Risk of Bias 2 (RoB 2) tool.</p></list-item><list-item><p>Diagnostic accuracy studies: Quality Assessment of Diagnostic Accuracy Studies-2 (QUADAS-2) tool.</p></list-item><list-item><p>Cross-sectional and cohort observational studies: Joanna Briggs Institute (JBI) checklist for analytical studies.</p></list-item><list-item><p>Qualitative studies: JBI checklist for qualitative research.</p></list-item><list-item><p>Mixed methods studies: Mixed Methods Appraisal Tool (MMAT).</p></list-item></list><p>Each study was classified into categories of low risk, moderate risk, or high risk in the evaluated domains. The results were graphically synthesized using stacked bar charts to facilitate the global interpretation of methodological quality.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Selection</title><p>The initial search identified 583 records in the selected databases. After removing 99 duplicates, 484 titles and abstracts were evaluated, of which 412 were excluded for not meeting the inclusion criteria. Thirty-two full texts were reviewed, excluding 14 studies for the following reasons:</p><list list-type="bullet"><list-item><p>Noneligible study type (n=10).</p></list-item><list-item><p>Diagnosis unrelated to the research question (n=4).</p></list-item></list><p>Finally, 18 studies met the criteria and were included in the qualitative synthesis. The complete process is shown in the PRISMA diagram (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study flowchart (produced according to the PRISMA [Preferred Reporting Items for Systematic Reviews and Meta-Analyses] 2020 flow diagram). <sup>a</sup>Consider, if feasible, reporting the number of records identified from each database or register searched (rather than the total number across all databases/registers). <sup>b</sup>If automation tools were used, indicate how many records were excluded by a human and how many were excluded by automation tools.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e93672_fig01.png"/></fig></sec><sec id="s3-2"><title>Study Characteristics</title><p>The 18 included studies were published between 2021 and 2025, covering diverse geographical contexts (North America, Europe, Asia, and Oceania) and samples ranging from 20 to 3902 participants. Most studies were conducted in clinical, community, or university settings, with adult populations presenting with depressive disorders, anxiety, suicide risk, cognitive impairment, schizophrenia, or PTSD.</p><p>Regarding interventions, 2 major categories were identified:</p><list list-type="bullet"><list-item><p>Rule-based chatbots are designed to administer psychometric scales or screening protocols.</p></list-item><list-item><p>Generative models (LLMs), primarily ChatGPT (versions 3.5, 4, and 4o), along with other models such as Claude, Gemini, Llama, and commercial chatbots (eg, Woebot and Vickybot).</p></list-item></list><p>The most commonly used reference instruments were the Patient Health Questionnaire-9 (PHQ-9), Generalized Anxiety Disorder-7 Scale (GAD-7), Social Phobia Inventory (SPIN), suicide risk scales, including the Suicidal Ideation Response Inventory-2 (SIRI-2), structured diagnostic interviews, and cognitive tests. Methodological designs included RCTs, validation studies, cross-sectional studies, qualitative studies, and feasibility pilots, reflecting notable heterogeneity in objectives and metrics.</p></sec><sec id="s3-3"><title>Outcomes</title><p>The findings were grouped into 3 dimensions:</p><sec id="s3-3-1"><title>Type of Chatbot and Assessment Modality</title><p>On the one hand, rule-based chatbots showed high reliability in administering psychometric scales (Cronbach &#x03B1; &#x003E;0.85) and good acceptability (Acceptability of Intervention Measure [AIM] scores &#x003E;19/25). On the other hand, generative models (LLMs) achieved high diagnostic accuracy metrics in screening and classification tasks, with sensitivities between 0.84 and 0.93 and specificities between 0.80 and 0.96 for depression and anxiety (<xref ref-type="fig" rid="figure2">Figure 2</xref> [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Forest plot of sensitivity (S) by study [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. PTSD: posttraumatic stress disorder.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e93672_fig02.png"/></fig></sec><sec id="s3-3-2"><title>Type of Assessment (Screening vs Diagnosis)</title><p>Regarding screening, chatbots achieved equivalence with traditional methods on scales such as the PHQ-9 and GAD-7, with intraclass correlation coefficients of 0.70 to 0.80. In diagnosis tasks, LLMs showed performance comparable to professionals in complex tasks (eg, suicide risk and schizophrenia), with correlations with experts up to <italic>r</italic>=0.96 (<xref ref-type="fig" rid="figure3">Figure 3</xref> [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Forest plot of specificity (E) by study [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. PTSD: posttraumatic stress disorder.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e93672_fig03.png"/></fig></sec><sec id="s3-3-3"><title>Type of Disorder</title><p>The greatest volume of evidence was found for depression and anxiety, with an AUC between 0.80 and 0.84 in adjusted models. Suicide risk assessment demonstrated a high concordance with experts (<italic>r</italic>&#x003E;0.90) in clinical vignettes. Promising results were found for cognitive impairment and schizophrenia, especially when combining LLMs with machine learning (ML) techniques (<italic>F</italic><sub>1</sub>-score &#x003E;0.90 in some studies; <xref ref-type="fig" rid="figure4">Figure 4</xref> [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]).</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Forest plot of area under the curve (AUC) by study [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]. PTSD: posttraumatic stress disorder.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e93672_fig04.png"/></fig></sec><sec id="s3-3-4"><title>Acceptability and Usability</title><p>Most studies reported high satisfaction (&#x2265;70% of users) and good intention for sustained use, although barriers related to cognitive load and technological familiarity were identified in older populations.</p><p>Forest plots were created for sensitivity (S), specificity (E), AUC, <italic>F</italic><sub>1</sub>-score, and accuracy, based on point estimates reported in the included studies. Since most articles did not report CIs or study weights, the figures show normalized point values (0&#x2010;1) with reference lines at 0.80 and 0.90. Overall, the results indicate high performance in language-based screening tasks (eg, depression, anxiety, and suicide risk), and high accuracies in hybrid approaches (LLM/embeddings+ML) for cognitive impairment; however, variability between conditions and designs is observed, which warrants cautious interpretations and the need for standardized reporting (metrics with CI and weighting by sample size).</p></sec></sec><sec id="s3-4"><title>Use of a Chatbot for Diagnosis/Assessment (Through Direct Interaction)</title><sec id="s3-4-1"><title>Cognitive Impairment in Free Dialogue</title><p>In a conversational application, de Arriba-P&#x00E9;rez et al [<xref ref-type="bibr" rid="ref20">20</xref>] compared 3 approaches: n-grams, a language model as a direct classifier, and a hybrid flow in which the model extracts language representations (features) and a random forest performs the classification. The hybrid flow achieved an accuracy of 98.47%, with macro averages close to 98% and sensitivity for the &#x201C;impairment&#x201D; class of 97.78%, clearly surpassing n-grams (76.67%) and the direct classifier (57.17%&#x2010;61.19%). These data show that conversational interaction combined with task-trained processing achieves very high performance (<xref ref-type="fig" rid="figure5">Figure 5</xref> [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]).</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Forest plot of <italic>F</italic><sub>1</sub>-score by study [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]. PTSD: posttraumatic stress disorder.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e93672_fig05.png"/></fig></sec><sec id="s3-4-2"><title>Chatbot-Guided Interview Combined With Neuroimaging</title><p>In a multimodal approach, Li [<xref ref-type="bibr" rid="ref47">47</xref>] integrated a ChatGPT-guided interview with functional magnetic resonance imaging (fMRI) to classify psychiatric diagnoses in adults with a confirmed single diagnosis. The complete model achieved an accuracy of 85.7% in an internal test and 83.5% in external validation (n=100), with an <italic>F</italic><sub>1</sub>-score of 85.5%. Ablation analyzes showed that the linguistic component alone reaches approximately 83% and that combining it with fMRI raises performance to approximately 87%. Compared with the study by de Arriba-P&#x00E9;rez et al [<xref ref-type="bibr" rid="ref20">20</xref>], the result is somewhat lower, consistent with the greater complexity of multidiagnostic and multimodal classification.</p><p>In binary dialogue-based evaluation (impairment yes/no), figures close to 98% are reached [<xref ref-type="bibr" rid="ref20">20</xref>]; in multidiagnostic classification guided by conversation and combined with neuroimaging, accuracy is around 85% to 84% [<xref ref-type="bibr" rid="ref47">47</xref>] (<xref ref-type="fig" rid="figure6">Figure 6</xref> [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]).</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Forest plot of accuracy/precision by study [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]. PTSD: posttraumatic stress disorder.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e93672_fig06.png"/></fig></sec></sec><sec id="s3-5"><title>Use Language Models to Process Data (Without Direct Interaction)</title><sec id="s3-5-1"><title>Brief Clinical Text, Interviews, and Narratives</title><sec id="s3-5-1-1"><title>Depression and Suicide Risk in Short Clinical Narratives</title><p>In hospitalized patients, Lho et al [<xref ref-type="bibr" rid="ref41">41</xref>] analyzed text from the incomplete sentences test. The model&#x2019;s performance without examples was moderate (AUC 0.720 for depression; 0.731 for suicide risk), with improvement when providing a few examples for depression (0.754). The best result was offered by the combination of language representations+XGBoost, with AUC 0.841 (depression) and 0.724 (suicide); self-concept was the most informative textual domain.</p></sec><sec id="s3-5-1-2"><title>Depression in Spontaneous Daily Writing</title><p>Shin et al [<xref ref-type="bibr" rid="ref44">44</xref>] showed accuracy 0.902 and specificity 0.955 in personal diaries after model fine-tuning; without fine-tuning, the balanced accuracy reached 0.844 with sensitivity 0.929, supporting the use of ecological language for screening when the model is optimized.</p></sec><sec id="s3-5-1-3"><title>Social Anxiety in Semistructured Interview</title><p>Ohse et al [<xref ref-type="bibr" rid="ref43">43</xref>] found high convergent validity between model-estimated severity and SPIN (<italic>r</italic>=0.79) and a &#x201C;probable&#x201D; case classification with <italic>F</italic><sub>1</sub>-score of 0.84. In clinical interview tasks, these figures are high.</p></sec><sec id="s3-5-1-4"><title>Schizophrenia: Language and Thought</title><p>With semantic connectivity in the interview, Voppel et al [<xref ref-type="bibr" rid="ref23">23</xref>] obtained an accuracy of 85%, sensitivity of 86%, and specificity of 84% (cross-validation). Pugh et al [<xref ref-type="bibr" rid="ref24">24</xref>] showed that language models (GPT-3.5 or 4, Llama-3) match experts when scoring coherence, tangentiality, and content, although with variability between runs that improves when adjusting parameters and aggregating outputs. Overall, the linguistic signal for schizophrenia is robust when the method is stable.</p></sec><sec id="s3-5-1-5"><title>Postpartum PTSD From Birth Narratives</title><p>Bartal et al [<xref ref-type="bibr" rid="ref25">25</xref>] compared generative approaches with a classifier trained on language representations: AUC=0.80, <italic>F</italic><sub>1</sub>-score=0.81, sensitivity=0.85, specificity=0.75, with clear superiority over &#x201C;no fine-tuning&#x201D; use. Compared to depression in diaries [<xref ref-type="bibr" rid="ref44">44</xref>] and depression in clinical narratives [<xref ref-type="bibr" rid="ref41">41</xref>], performance is similar (AUC&#x2248;0.80&#x2010;0.84).</p></sec><sec id="s3-5-1-6"><title>Symptom Extraction and Clinical Summary</title><p>So et al [<xref ref-type="bibr" rid="ref46">46</xref>] aligned a model to label symptoms in interviews, achieving <italic>F</italic><sub>1</sub>-score of 0.82 and well-valued summaries in coherence and consistency. This support can structure interviews and save documentation time.</p></sec><sec id="s3-5-1-7"><title>Multimodal Classification Based on Language+fMRI (Without Direct Conversation)</title><p>Li [<xref ref-type="bibr" rid="ref47">47</xref>] showed that language alone already provides approximately 83% accuracy and that adding fMRI raises that value to approximately 87%, confirming the incremental value of combining text with biomarkers.</p></sec><sec id="s3-5-1-8"><title>Electronic Health Records</title><p>Detection of cognitive impairment in electronic health records (EHRs; without keyword filters).</p><p>Du et al [<xref ref-type="bibr" rid="ref12">12</xref>] developed an ensemble (generative model+attention neural network+Extreme Gradient Boosting [XGBoost]) that achieved <italic>F</italic><sub>1</sub>-score of 92.1%, sensitivity of 94.2%, precision of 90.2%, and specificity of 99.6%, surpassing each component separately (the generative model, optimized by instructions, reached <italic>F</italic><sub>1</sub>-score of 80.3%). Compared to studies with narratives or diaries, the EHR offers larger samples and, with the ensemble, very high metrics.</p></sec></sec></sec><sec id="s3-6"><title>Vignettes and Other Contexts</title><sec id="s3-6-1"><title>Judgment Before Suicidal Ideation (SIRI-2 Vignettes)</title><p>McBain et al [<xref ref-type="bibr" rid="ref21">21</xref>] observed high correlations among suicidologists (<italic>r</italic>=0.96; 0.93; 0.81) and elevated reliability, although with a leniency bias; Shinan-Altman et al [<xref ref-type="bibr" rid="ref22">22</xref>] showed that depression and access to weapons elevate risk assessment, a more consistent pattern in the more advanced version of the model.</p></sec><sec id="s3-6-2"><title>Prediction of Symptom Changes</title><p>Hur et al [<xref ref-type="bibr" rid="ref37">37</xref>] showed that linguistic sentiment modeled by a language model predicts worsening of depression at 3 weeks, with performance similar to human evaluators.</p></sec><sec id="s3-6-3"><title>Other Comparative Areas</title><p>Jin et al [<xref ref-type="bibr" rid="ref7">7</xref>] observed improvements when requesting explicit reasoning, with failures in atypical presentations; evidence indicates that AI tools for mental health screening may show differential performance across demographic subgroups, highlighting the need for systematic bias assessment [<xref ref-type="bibr" rid="ref30">30</xref>].</p></sec></sec><sec id="s3-7"><title>Risk of Bias</title><p>Risk of bias analysis was conducted applying specific tools according to the methodological design of each study. For RCTs, the Cochrane RoB 2 tool was used, while diagnostic accuracy studies were evaluated using QUADAS-2. Qualitative studies were analyzed with the JBI checklist for qualitative research, and mixed methods studies with the MMAT. Finally, cross-sectional studies were evaluated with the JBI checklist for descriptive studies. Each study was classified according to the risk of bias in the different methodological domains, assigning judgments of &#x201C;low risk,&#x201D; &#x201C;moderate risk,&#x201D; or &#x201C;high risk.&#x201D;</p><p>The results were visualized using stacked bar charts by domain, facilitating the global interpretation of the methodological quality of the included evidence. Cross-sectional studies presented a higher proportion of moderate bias in sampling strategies and measurement validity. In RCTs, the most frequent bias was observed in the blinding of participants and personnel. Studies evaluated with QUADAS-2 showed moderate bias in participant selection and intervention classification.</p><p>Additional per-study risk-of-bias visualizations are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. Overall, the included studies present heterogeneous methodological quality, with a predominance of moderate bias in critical dimensions of internal validity. This methodological variability should be considered when interpreting the results of efficacy and accuracy of chatbots in psychological assessment.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This systematic review demonstrates that chatbots and AI-based conversational agents achieve clinically relevant performance in mental health screening and assessment. Rule-based systems showed high reliability (Cronbach &#x03B1;&#x003E;0.85) in administering standardized scales, while generative LLMs achieved sensitivities of 0.84&#x2010;0.93 and specificities of 0.80&#x2010;0.96 for depression and anxiety. LLMs demonstrated strong correlations with expert clinicians (<italic>r</italic> values up to 0.96) in suicide risk assessment and exceptional performance when combined with ML (<italic>F</italic><sub>1</sub>-score=92.1%) in cognitive impairment detection [<xref ref-type="bibr" rid="ref12">12</xref>]. User acceptability was generally high (&#x2265;70%), although barriers existed for older adults and those with lower digital literacy. These findings suggest AI-driven tools can replicate traditional psychometric protocols while offering additional capabilities through natural language processing; however, high-performance metrics must be interpreted cautiously given methodological heterogeneity and moderate risk of bias.</p><p>Our findings extend beyond previous systematic reviews, whereas those reviews [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>] reported promising but limited evidence for chatbot effectiveness in mental health contexts. These reviews focused primarily on therapeutic interventions rather than assessment capabilities. A critical distinction in our review is the emphasis on generative LLMs, which represent a qualitative advance from earlier rule-based systems. Our synthesis captures the rapid evolution from 2023 to 2025, during which LLM-based applications surged from 16% to 45% in mental health chatbot studies [<xref ref-type="bibr" rid="ref15">15</xref>]. Recent evidence demonstrates that GPT-4 and similar models can pass psychiatric licensing examinations and perform diagnostic reasoning comparable to experienced clinicians [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>], capabilities not observed in earlier generation systems. Our findings on cognitive impairment detection [<xref ref-type="bibr" rid="ref12">12</xref>] and suicide risk assessment [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref27">27</xref>] extend the evidence base beyond the depression and anxiety focus of prior reviews.</p></sec><sec id="s4-2"><title>Strengths and Limitations of the Evidence Base</title><p>Included studies used appropriate reference standards, including validated instruments (PHQ-9 and GAD-7) and structured clinical interviews, with several using external validation cohorts. However, significant limitations imply careful interpretation. Risk of bias assessment revealed moderate to high risk in critical domains, particularly participant selection, blinding procedures, and selective reporting. Sample sizes varied considerably (20-3902 participants), and most used cross-sectional designs, precluding longitudinal assessment. Importantly, only 47% of studies focused on clinical efficacy testing, with most LLM-based studies (77%) remaining in early validation phases [<xref ref-type="bibr" rid="ref15">15</xref>]. Technological advancement has progressed faster than the multistage validation processes necessary for robust clinical evidence, resulting in limited data on long-term effectiveness and real-world implementation. Heterogeneity in reported metrics and a lack of standardized evaluation frameworks complicated cross-study comparisons and precluded meta-analysis for most outcomes.</p></sec><sec id="s4-3"><title>Clinical Implications</title><p>The evidence suggests several potential applications for AI-driven assessment tools. First, chatbots administering standardized instruments demonstrate equivalence with traditional modes, offering opportunities for remote screening, automated triage, and population scrutiny [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. This could help address treatment gaps in underserved areas and reduce waiting times for initial assessment. Second, LLMs show promise as clinical decision support tools for extracting structured information from clinical narratives [<xref ref-type="bibr" rid="ref46">46</xref>], identifying at-risk individuals requiring urgent evaluation [<xref ref-type="bibr" rid="ref27">27</xref>], and augmenting cognitive assessment in resource-limited settings [<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>However, safe implementation requires clear clinical protocols, professional oversight, and integration with existing health care workflows. AI should function as a complementary tool that enhances rather than replaces clinical judgment. Hybrid models combining automated screening with human review for positive or uncertain cases may optimize sensitivity while maintaining specificity. EHR integration is essential for continuity of care, but it requires attention to data security, patient consent, and algorithmic transparency. Critically, the absence of longitudinal studies limits conclusions about sustained accuracy, calibration drift, or adaptation to changing diagnostic criteria.</p></sec><sec id="s4-4"><title>Algorithmic Bias and Health Equity</title><p>A major concern emerging from this review is the potential for AI systems to perpetuate or amplify health disparities. Empirical work shows that performance and risk estimates can vary across demographic subgroups, underscoring the importance of bias evaluation and mitigation [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. Adler et al [<xref ref-type="bibr" rid="ref30">30</xref>] demonstrated that sensed-behavioral patterns predictive of depression vary substantially across demographic subgroups, with AI tools incorrectly ranking individuals with depression from certain groups as lower risk than healthier individuals from other groups. Qualitative analyses reveal that race-explicit or race-implied patient information frequently results in inferior treatment recommendations from LLMs [<xref ref-type="bibr" rid="ref7">7</xref>], and natural language processing models exhibit measurable biases related to religion, race, gender, nationality, sexuality, and age [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>].</p><p>Moreover, acceptability findings suggest differential engagement by age, education, and technological familiarity, with older adults and those with lower digital literacy reporting lower acceptability [<xref ref-type="bibr" rid="ref48">48</xref>], potentially expanding access gaps for vulnerable populations. Addressing these inequities requires diverse and representative training datasets, rigorous bias testing across protected characteristics, transparency in algorithmic decision-making, human oversight mechanisms, and continuous monitoring of real-world performance stratified by demographic factors.</p></sec><sec id="s4-5"><title>Ethical Considerations and Implementation Challenges</title><p>Beyond bias, several ethical concerns require attention. The &#x201C;black box&#x201D; nature of many LLMs raises questions about accountability when errors occur [<xref ref-type="bibr" rid="ref49">49</xref>]. Clear governance frameworks delineating roles and liabilities are essential. Informed consent presents unique challenges, as patients must understand when AI is involved in their assessment, how their data will be used, and the limitations of algorithmic predictions. Data privacy and security are paramount, given the sensitivity of mental health information. The risk of overreliance on automated systems could lead to deskilling of clinicians or premature diagnostic closure. AI should augment, not replace, comprehensive clinical assessment that considers contextual factors, patient preferences, and therapeutic relationships. Implementation also faces practical barriers, including EHR integration, evolving regulatory frameworks, underdeveloped reimbursement mechanisms, and the need for clinician training.</p></sec><sec id="s4-6"><title>Future Directions</title><p>Several critical gaps require attention. First, prospective longitudinal studies are needed to evaluate sustained performance, calibration over time, and clinical utility in real-world settings. Second, comparative effectiveness research should directly compare AI-driven assessment against standard care in pragmatic trials, measuring diagnostic accuracy, access, efficiency, and clinical outcomes. Third, research must systematically evaluate performance across diverse populations with prespecified, adequately powered subgroup analyses. Fourth, studies should explore optimal human-AI collaboration models and the required clinician training. Fifth, research on algorithmic fairness and bias mitigation strategies is essential. Sixth, implementation science research should identify facilitators, barriers, and sustainment strategies. Finally, regulatory science research is needed to establish appropriate evaluation frameworks and postmarket surveillance requirements.</p></sec><sec id="s4-7"><title>Implications for Policy and Practice</title><p>Health care systems considering AI adoption should prioritize infrastructure for secure data management, clinical protocols specifying appropriate use cases and oversight requirements, clinician training programs, bias monitoring mechanisms, and patient engagement. Regulatory bodies should develop clear approval pathways requiring evidence of clinical validity, algorithmic transparency, bias testing, and ongoing surveillance. Professional organizations should establish clinical practice guidelines for AI-assisted assessment, addressing appropriate use cases, limitations, documentation requirements, and ethical obligations. Researchers and developers should prioritize open science practices, including sharing algorithms and evaluation frameworks where ethically permissible. Policymakers should ensure that AI deployment does not exacerbate existing disparities and that vulnerable populations maintain access to human-delivered care.</p></sec><sec id="s4-8"><title>Conclusions</title><p>AI-driven conversational agents demonstrate clinically relevant performance in mental health screening and assessment, with potential to expand access and to support clinical decision-making. However, realizing this potential requires addressing substantial methodological, ethical, and practical challenges. The evidence base remains limited by heterogeneity, short-term follow-up, and moderate risk of bias. Algorithmic bias poses serious threats to health equity that demand proactive mitigation. Implementation requires careful attention to clinical integration, human oversight, transparency, and patient engagement. The goal should not be to replace human clinicians but to develop complementary tools that enhance efficiency, access, and quality while preserving the essential human elements of empathy, contextual understanding, and ethical judgment. A collaborative model integrating AI capabilities with human expertise may offer a path toward more accessible, timely, and equitable mental health assessment. However, this vision requires sustained research, thoughtful regulation, and a commitment to equity as technological capabilities continue to evolve.</p></sec></sec></body><back><ack><p>We used ChatGPT (OpenAI) to assist with language refinement/editing of the manuscript and to refine/format the code reported in the manuscript to improve readability without altering its functionality. The authors reviewed, edited, and verified all AI-assisted outputs and remain fully accountable for the content.</p></ack><notes><sec><title>Funding</title><p>This work received funding from the International University of La Rioja (UNIR) and the Foundation for Biomedical Research and Innovation of the Infanta Leonor and Southeast University Hospitals.SD holds a Torres Quevedo postdoctoral fellowship (ref. PTQ2024-013768).</p></sec><sec><title>Data Availability</title><p>All data extracted and analyzed in this systematic review are derived from published studies. The data supporting the findings of this review are available within the article and its Multimedia Appendices.</p></sec></notes><fn-group><fn fn-type="con"><p>Formal analysis: JMG-C</p><p>Writing &#x2013; original draft: IM-E, SD, IMG, SA-G, RF, JQ</p><p>Writing &#x2013; review &#x0026; editing: SD, IMG, SA-G, RF, JMG-C, M&#x00C1;&#x00C1;-M, JQ</p></fn><fn fn-type="conflict"><p>RF disclosed her role as a cofounder of a digital psychological product company, stating that she is not currently receiving any economic compensation. SD reported being employed by the same company. IMG is professionally affiliated with MIYU, a digital mental health initiative, and holds phantom shares in the project; no remuneration or funding has been received in connection with this role. MIYU had no role in the funding, design, conduct, analysis, interpretation, or preparation of this systematic review. None of the chatbots evaluated in the review were developed, owned, or commercialized by MIYU. JQ has ongoing research projects in AI, holds stocks in Q&#x0026;G, and has a grant from the Instituto de Salud Carlos III (ISCIII). The other authors declare no conflicts of interest. Beyond the roles described above, the authors have no additional funding, remuneration, or equity relationships related to MIYU.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AIM</term><def><p>Acceptability of Intervention Measure</p></def></def-item><def-item><term id="abb2">AUC</term><def><p>area under the curve</p></def></def-item><def-item><term id="abb3">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb4">fMRI</term><def><p>functional magnetic resonance imaging</p></def></def-item><def-item><term id="abb5">GAD-7</term><def><p>Generalized Anxiety Disorder-7</p></def></def-item><def-item><term id="abb6">JBI</term><def><p>Joanna Briggs Institute</p></def></def-item><def-item><term id="abb7">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb8">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb9">MMAT</term><def><p>Mixed Methods Appraisal Tool</p></def></def-item><def-item><term id="abb10">PHQ-9</term><def><p>Patient Health Questionnaire-9</p></def></def-item><def-item><term id="abb11">PIO</term><def><p>Population, Intervention, Outcome</p></def></def-item><def-item><term id="abb12">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb13">PROSPERO</term><def><p>International Prospective Register of Systematic Reviews</p></def></def-item><def-item><term id="abb14">PTSD</term><def><p>posttraumatic stress disorder</p></def></def-item><def-item><term id="abb15">QUADAS-2</term><def><p>Quality Assessment of Diagnostic Accuracy Studies-2</p></def></def-item><def-item><term id="abb16">RCT</term><def><p>randomized controlled trial</p></def></def-item><def-item><term id="abb17">RoB 2</term><def><p>Risk of Bias 2</p></def></def-item><def-item><term id="abb18">SIRI-2</term><def><p>Suicidal Ideation Response Inventory&#x2013;Revised</p></def></def-item><def-item><term id="abb19">SPIN</term><def><p>Social Phobia Inventory</p></def></def-item><def-item><term id="abb20">XGBoost</term><def><p>Extreme Gradient Boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="report"><article-title>World mental health report: transforming mental health for all</article-title><year>2022</year><access-date>2026-01-30</access-date><publisher-name>World Health Organization</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://iris.who.int/server/api/core/bitstreams/40e5a13a-fe50-4efa-b56d-6e8cf00d5bfa/content">https://iris.who.int/server/api/core/bitstreams/40e5a13a-fe50-4efa-b56d-6e8cf00d5bfa/content</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>GBD 2019 Mental Disorders Collaborators</collab></person-group><article-title>Global, regional, and national burden of 12 mental disorders in 204 countries and territories, 1990-2019: a systematic analysis for the Global Burden of Disease Study 2019</article-title><source>Lancet Psychiatry</source><year>2022</year><month>02</month><volume>9</volume><issue>2</issue><fpage>137</fpage><lpage>150</lpage><pub-id pub-id-type="doi">10.1016/S2215-0366(21)00395-3</pub-id><pub-id pub-id-type="medline">35026139</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="report"><article-title>A new benchmark for mental health systems: tackling the social and economic costs of mental ill-health</article-title><year>2021</year><access-date>2026-08-12</access-date><publisher-name>OECD Publishing</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.oecd.org/content/dam/oecd/en/publications/reports/2021/06/a-new-benchmark-for-mental-health-systems_c0cce868/4ed890f6-en.pdf">https://www.oecd.org/content/dam/oecd/en/publications/reports/2021/06/a-new-benchmark-for-mental-health-systems_c0cce868/4ed890f6-en.pdf</ext-link></comment></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="report"><article-title>Communication from the Commission to the European Parliament, the Council, the European Economic and Social Committee and the Committee of the Regions on a comprehensive approach to mental health</article-title><year>2023</year><access-date>2026-08-12</access-date><publisher-name>European Commission</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://health.ec.europa.eu/document/download/cef45b6d-a871-44d5-9d62-3cecc47eda89_en?filename=com_2023_298_1_act_en.pdf">https://health.ec.europa.eu/document/download/cef45b6d-a871-44d5-9d62-3cecc47eda89_en?filename=com_2023_298_1_act_en.pdf</ext-link></comment></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thornicroft</surname><given-names>G</given-names> </name><name name-style="western"><surname>Sunkel</surname><given-names>C</given-names> </name><name name-style="western"><surname>Alikhon Aliev</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The Lancet Commission on ending stigma and discrimination in mental health</article-title><source>Lancet</source><year>2022</year><month>10</month><day>22</day><volume>400</volume><issue>10361</issue><fpage>1438</fpage><lpage>1480</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(22)01470-2</pub-id><pub-id pub-id-type="medline">36223799</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patel</surname><given-names>V</given-names> </name><name name-style="western"><surname>Saxena</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lund</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Transforming mental health systems globally: principles and policy recommendations</article-title><source>Lancet</source><year>2023</year><month>08</month><day>19</day><volume>402</volume><issue>10402</issue><fpage>656</fpage><lpage>666</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(23)00918-2</pub-id><pub-id pub-id-type="medline">37597892</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>P</given-names> </name><etal/></person-group><article-title>The applications of large language models in mental health: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>5</day><volume>27</volume><issue>1</issue><fpage>e69284</fpage><pub-id pub-id-type="doi">10.2196/69284</pub-id><pub-id pub-id-type="medline">40324177</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>KD</given-names> </name><name name-style="western"><surname>Fernandez</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Schwartz</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Comparing GPT-4 and human researchers in health care data analysis: qualitative description study</article-title><source>J Med Internet Res</source><year>2024</year><month>08</month><day>21</day><volume>26</volume><fpage>e56500</fpage><pub-id pub-id-type="doi">10.2196/56500</pub-id><pub-id pub-id-type="medline">39167785</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gaffney</surname><given-names>H</given-names> </name><name name-style="western"><surname>Mansell</surname><given-names>W</given-names> </name><name name-style="western"><surname>Tai</surname><given-names>S</given-names> </name></person-group><article-title>Conversational agents in the treatment of mental health problems: mixed-method systematic review</article-title><source>JMIR Ment Health</source><year>2019</year><month>10</month><day>18</day><volume>6</volume><issue>10</issue><fpage>e14166</fpage><pub-id pub-id-type="doi">10.2196/14166</pub-id><pub-id pub-id-type="medline">31628789</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abd-Alrazaq</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Rababeh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Alajlani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bewick</surname><given-names>BM</given-names> </name><name name-style="western"><surname>Househ</surname><given-names>M</given-names> </name></person-group><article-title>Effectiveness and safety of using chatbots to improve mental health: systematic review and meta-analysis</article-title><source>J Med Internet Res</source><year>2020</year><month>07</month><day>13</day><volume>22</volume><issue>7</issue><fpage>e16021</fpage><pub-id pub-id-type="doi">10.2196/16021</pub-id><pub-id pub-id-type="medline">32673216</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name></person-group><article-title>Evaluating diagnostic accuracy and treatment efficacy in mental health: a comparative analysis of large language model tools and mental health professionals</article-title><source>Eur J Investig Health Psychol Educ</source><year>2025</year><month>01</month><day>18</day><volume>15</volume><issue>1</issue><fpage>9</fpage><pub-id pub-id-type="doi">10.3390/ejihpe15010009</pub-id><pub-id pub-id-type="medline">39852192</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Du</surname><given-names>X</given-names> </name><name name-style="western"><surname>Novoa-Laurentiev</surname><given-names>J</given-names> </name><name name-style="western"><surname>Plasek</surname><given-names>JM</given-names> </name><etal/></person-group><article-title>Enhancing early detection of cognitive decline in the elderly: a comparative study utilizing large language models in clinical notes</article-title><source>EBioMedicine</source><year>2024</year><month>11</month><volume>109</volume><fpage>105401</fpage><pub-id pub-id-type="doi">10.1016/j.ebiom.2024.105401</pub-id><pub-id pub-id-type="medline">39396423</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Boucher</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Harake</surname><given-names>NR</given-names> </name><name name-style="western"><surname>Ward</surname><given-names>HE</given-names> </name><etal/></person-group><article-title>Artificially intelligent chatbots in digital mental health interventions: a review</article-title><source>Expert Rev Med Devices</source><year>2021</year><month>12</month><volume>18</volume><issue>sup1</issue><fpage>37</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1080/17434440.2021.2013200</pub-id><pub-id pub-id-type="medline">34872429</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mayor</surname><given-names>E</given-names> </name></person-group><article-title>Chatbots and mental health: a scoping review of reviews</article-title><source>Curr Psychol</source><year>2025</year><month>08</month><volume>44</volume><issue>15</issue><fpage>13619</fpage><lpage>13640</lpage><pub-id pub-id-type="doi">10.1007/s12144-025-08094-2</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hua</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Siddals</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Charting the evolution of artificial intelligence mental health chatbots from rule-based systems to large language models: a systematic review</article-title><source>World Psychiatry</source><year>2025</year><month>10</month><volume>24</volume><issue>3</issue><fpage>383</fpage><lpage>394</lpage><pub-id pub-id-type="doi">10.1002/wps.21352</pub-id><pub-id pub-id-type="medline">40948070</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cheng</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>WJ</given-names> </name><etal/></person-group><article-title>The now and future of ChatGPT and GPT in psychiatry</article-title><source>Psychiatry Clin Neurosci</source><year>2023</year><month>11</month><volume>77</volume><issue>11</issue><fpage>592</fpage><lpage>596</lpage><pub-id pub-id-type="doi">10.1111/pcn.13588</pub-id><pub-id pub-id-type="medline">37612880</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hanss</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sarma</surname><given-names>KV</given-names> </name><name name-style="western"><surname>Glowinski</surname><given-names>AL</given-names> </name><etal/></person-group><article-title>Assessing the accuracy and reliability of large language models in psychiatry using standardized multiple-choice questions: cross-sectional study</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>20</day><volume>27</volume><issue>1</issue><fpage>e69910</fpage><pub-id pub-id-type="doi">10.2196/69910</pub-id><pub-id pub-id-type="medline">40392576</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Evaluation of large language models on mental health: from knowledge test to illness diagnosis</article-title><source>Front Psychiatry</source><year>2025</year><volume>16</volume><fpage>1646974</fpage><pub-id pub-id-type="doi">10.3389/fpsyt.2025.1646974</pub-id><pub-id pub-id-type="medline">40842952</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Tian</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ho</surname><given-names>GWK</given-names> </name><name name-style="western"><surname>Yorke</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hui</surname><given-names>V</given-names> </name></person-group><article-title>The effectiveness of AI chatbots in alleviating mental distress and promoting health behaviors among adolescents and young adults: systematic review and meta-analysis</article-title><source>J Med Internet Res</source><year>2025</year><month>11</month><day>26</day><volume>27</volume><issue>1</issue><fpage>e79850</fpage><pub-id pub-id-type="doi">10.2196/79850</pub-id><pub-id pub-id-type="medline">41313175</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Arriba-P&#x00E9;rez</surname><given-names>F</given-names> </name><name name-style="western"><surname>Garc&#x00ED;a-M&#x00E9;ndez</surname><given-names>S</given-names> </name><name name-style="western"><surname>Otero-Mosquera</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gonz&#x00E1;lez-Casta&#x00F1;o</surname><given-names>FJ</given-names> </name></person-group><article-title>Explainable cognitive decline detection in free dialogues with a Machine Learning approach based on pre-trained large language models</article-title><source>Appl Intell</source><year>2024</year><month>12</month><volume>54</volume><issue>24</issue><fpage>12613</fpage><lpage>12628</lpage><pub-id pub-id-type="doi">10.1007/s10489-024-05808-0</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McBain</surname><given-names>RK</given-names> </name><name name-style="western"><surname>Cantor</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>LA</given-names> </name><etal/></person-group><article-title>Competency of large language models in evaluating appropriate responses to suicidal ideation: comparative study</article-title><source>J Med Internet Res</source><year>2025</year><month>03</month><day>5</day><volume>27</volume><issue>1</issue><fpage>e67891</fpage><pub-id pub-id-type="doi">10.2196/67891</pub-id><pub-id pub-id-type="medline">40053817</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shinan-Altman</surname><given-names>S</given-names> </name><name name-style="western"><surname>Elyoseph</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name></person-group><article-title>The impact of history of depression and access to weapons on suicide risk assessment: a comparison of ChatGPT-3.5 and ChatGPT-4</article-title><source>PeerJ</source><year>2024</year><volume>12</volume><fpage>e17468</fpage><pub-id pub-id-type="doi">10.7717/peerj.17468</pub-id><pub-id pub-id-type="medline">38827287</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Voppel</surname><given-names>AE</given-names> </name><name name-style="western"><surname>de Boer</surname><given-names>JN</given-names> </name><name name-style="western"><surname>Brederoo</surname><given-names>SG</given-names> </name><name name-style="western"><surname>Schnack</surname><given-names>HG</given-names> </name><name name-style="western"><surname>Sommer</surname><given-names>I</given-names> </name></person-group><article-title>Quantified language connectedness in schizophrenia-spectrum disorders</article-title><source>Psychiatry Res</source><year>2021</year><month>10</month><volume>304</volume><fpage>114130</fpage><pub-id pub-id-type="doi">10.1016/j.psychres.2021.114130</pub-id><pub-id pub-id-type="medline">34332431</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pugh</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Chandler</surname><given-names>C</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Diaz-Asper</surname><given-names>C</given-names> </name><name name-style="western"><surname>Elvev&#x00E5;g</surname><given-names>B</given-names> </name><name name-style="western"><surname>Foltz</surname><given-names>PW</given-names> </name></person-group><article-title>Assessing dimensions of thought disorder with large language models: the tradeoff of accuracy and consistency</article-title><source>Psychiatry Res</source><year>2024</year><month>11</month><volume>341</volume><fpage>116119</fpage><pub-id pub-id-type="doi">10.1016/j.psychres.2024.116119</pub-id><pub-id pub-id-type="medline">39226873</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bartal</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jagodnik</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Dekel</surname><given-names>S</given-names> </name></person-group><article-title>AI and narrative embeddings detect PTSD following childbirth via birth stories</article-title><source>Sci Rep</source><year>2024</year><month>04</month><day>11</day><volume>14</volume><issue>1</issue><fpage>8336</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-54242-2</pub-id><pub-id pub-id-type="medline">38605073</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elyoseph</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Levkovich</surname><given-names>I</given-names> </name></person-group><article-title>Beyond human expertise: the promise and limitations of ChatGPT in suicide risk assessment</article-title><source>Front Psychiatry</source><year>2023</year><volume>14</volume><fpage>1213141</fpage><pub-id pub-id-type="doi">10.3389/fpsyt.2023.1213141</pub-id><pub-id pub-id-type="medline">37593450</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mohebbi</surname><given-names>M</given-names> </name><name name-style="western"><surname>O&#x2019;Callaghan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Winsberg</surname><given-names>M</given-names> </name></person-group><article-title>Large language models versus expert clinicians in crisis prediction among telemental health patients: comparative study</article-title><source>JMIR Ment Health</source><year>2024</year><month>08</month><day>2</day><volume>11</volume><fpage>e58129</fpage><pub-id pub-id-type="doi">10.2196/58129</pub-id><pub-id pub-id-type="medline">38876484</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Timmons</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Duong</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Simo Fiallo</surname><given-names>N</given-names> </name><etal/></person-group><article-title>A call to action on assessing and mitigating bias in artificial intelligence applications for mental health</article-title><source>Perspect Psychol Sci</source><year>2023</year><month>09</month><volume>18</volume><issue>5</issue><fpage>1062</fpage><lpage>1096</lpage><pub-id pub-id-type="doi">10.1177/17456916221134490</pub-id><pub-id pub-id-type="medline">36490369</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tavory</surname><given-names>T</given-names> </name></person-group><article-title>Regulating AI in mental health: ethics of care perspective</article-title><source>JMIR Ment Health</source><year>2024</year><month>09</month><day>19</day><volume>11</volume><fpage>e58493</fpage><pub-id pub-id-type="doi">10.2196/58493</pub-id><pub-id pub-id-type="medline">39298759</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Adler</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Stamatis</surname><given-names>CA</given-names> </name><name name-style="western"><surname>Meyerhoff</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Measuring algorithmic bias to analyze the reliability of AI tools that predict depression risk using smartphone sensed-behavioral data</article-title><source>Npj Ment Health Res</source><year>2024</year><month>04</month><day>22</day><volume>3</volume><issue>1</issue><fpage>17</fpage><pub-id pub-id-type="doi">10.1038/s44184-024-00057-y</pub-id><pub-id pub-id-type="medline">38649446</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Straw</surname><given-names>I</given-names> </name><name name-style="western"><surname>Callison-Burch</surname><given-names>C</given-names> </name></person-group><article-title>Artificial Intelligence in mental health and the biases of language based models</article-title><source>PLoS One</source><year>2020</year><volume>15</volume><issue>12</issue><fpage>e0240376</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0240376</pub-id><pub-id pub-id-type="medline">33332380</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>M</given-names> </name><name name-style="western"><surname>El-Attar</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Chaspari</surname><given-names>T</given-names> </name></person-group><article-title>Deconstructing demographic bias in speech-based machine learning models for digital health</article-title><source>Front Digit Health</source><year>2024</year><volume>6</volume><fpage>1351637</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2024.1351637</pub-id><pub-id pub-id-type="medline">39119589</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bailey</surname><given-names>RK</given-names> </name><name name-style="western"><surname>Mokonogho</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>A</given-names> </name></person-group><article-title>Racial and ethnic differences in depression: current perspectives</article-title><source>Neuropsychiatr Dis Treat</source><year>2019</year><volume>15</volume><fpage>603</fpage><lpage>609</lpage><pub-id pub-id-type="doi">10.2147/NDT.S128584</pub-id><pub-id pub-id-type="medline">30863081</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Anmella</surname><given-names>G</given-names> </name><name name-style="western"><surname>Sanabra</surname><given-names>M</given-names> </name><name name-style="western"><surname>Prim&#x00E9;-Tous</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Vickybot, a chatbot for anxiety-depressive symptoms and work-related burnout in primary care and health care professionals: development, feasibility, and potential effectiveness studies</article-title><source>J Med Internet Res</source><year>2023</year><month>04</month><day>3</day><volume>25</volume><fpage>e43293</fpage><pub-id pub-id-type="doi">10.2196/43293</pub-id><pub-id pub-id-type="medline">36719325</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dosovitsky</surname><given-names>G</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bunge</surname><given-names>EL</given-names> </name></person-group><article-title>Psychometric properties of a chatbot version of the PHQ-9 with adults and older adults</article-title><source>Front Digit Health</source><year>2021</year><volume>3</volume><fpage>645805</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2021.645805</pub-id><pub-id pub-id-type="medline">34713116</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>Q</given-names> </name></person-group><article-title>Physician versus large language model chatbot responses to web-based questions from autistic patients in Chinese: cross-sectional comparative analysis</article-title><source>J Med Internet Res</source><year>2024</year><month>04</month><day>30</day><volume>26</volume><issue>1</issue><fpage>e54706</fpage><pub-id pub-id-type="doi">10.2196/54706</pub-id><pub-id pub-id-type="medline">38687566</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hur</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Heffner</surname><given-names>J</given-names> </name><name name-style="western"><surname>Feng</surname><given-names>GW</given-names> </name><name name-style="western"><surname>Joormann</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rutledge</surname><given-names>RB</given-names> </name></person-group><article-title>Language sentiment predicts changes in depressive symptoms</article-title><source>Proc Natl Acad Sci U S A</source><year>2024</year><month>09</month><day>24</day><volume>121</volume><issue>39</issue><fpage>e2321321121</fpage><pub-id pub-id-type="doi">10.1073/pnas.2321321121</pub-id><pub-id pub-id-type="medline">39284070</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Hong</surname><given-names>M</given-names> </name></person-group><article-title>Development and evaluation of a mental health chatbot using ChatGPT 4.0: mixed methods user experience study with Korean users</article-title><source>JMIR Med Inform</source><year>2025</year><month>01</month><day>3</day><volume>13</volume><fpage>e63538</fpage><pub-id pub-id-type="doi">10.2196/63538</pub-id><pub-id pub-id-type="medline">39752663</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaywan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ibaida</surname><given-names>A</given-names> </name><name name-style="western"><surname>Miao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gu</surname><given-names>B</given-names> </name></person-group><article-title>Early detection of depression using a conversational AI bot: a non-clinical trial</article-title><source>PLoS One</source><year>2023</year><volume>18</volume><issue>2</issue><fpage>e0279743</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0279743</pub-id><pub-id pub-id-type="medline">36735701</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kosyluk</surname><given-names>K</given-names> </name><name name-style="western"><surname>Baeder</surname><given-names>T</given-names> </name><name name-style="western"><surname>Greene</surname><given-names>KY</given-names> </name><etal/></person-group><article-title>Mental distress, label avoidance, and use of a mental health chatbot: results from a US survey</article-title><source>JMIR Form Res</source><year>2024</year><month>04</month><day>12</day><volume>8</volume><issue>1</issue><fpage>e45959</fpage><pub-id pub-id-type="doi">10.2196/45959</pub-id><pub-id pub-id-type="medline">38607665</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lho</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Park</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Large language models and text embeddings for detecting depression and suicide in patient narratives</article-title><source>JAMA Netw Open</source><year>2025</year><month>05</month><day>1</day><volume>8</volume><issue>5</issue><fpage>e2511922</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.11922</pub-id><pub-id pub-id-type="medline">40408109</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tong</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Evaluating the agreement between ChatGPT-4 and validated questionnaires in screening for anxiety and depression in college students: a cross-sectional study</article-title><source>BMC Psychiatry</source><year>2025</year><month>04</month><day>10</day><volume>25</volume><issue>1</issue><fpage>359</fpage><pub-id pub-id-type="doi">10.1186/s12888-025-06798-0</pub-id><pub-id pub-id-type="medline">40211256</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ohse</surname><given-names>J</given-names> </name><name name-style="western"><surname>Had&#x017E;i&#x0107;</surname><given-names>B</given-names> </name><name name-style="western"><surname>Mohammed</surname><given-names>P</given-names> </name><etal/></person-group><article-title>GPT-4 shows potential for identifying social anxiety from clinical interview data</article-title><source>Sci Rep</source><year>2024</year><month>12</month><day>16</day><volume>14</volume><issue>1</issue><fpage>30498</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-82192-2</pub-id><pub-id pub-id-type="medline">39681627</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jung</surname><given-names>W</given-names> </name></person-group><article-title>Using large language models to detect depression from user-generated diary text data as a novel approach in digital mental health screening: instrument validation study</article-title><source>J Med Internet Res</source><year>2024</year><month>09</month><day>18</day><volume>26</volume><issue>1</issue><fpage>e54617</fpage><pub-id pub-id-type="doi">10.2196/54617</pub-id><pub-id pub-id-type="medline">39292502</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levi-Belz</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gamliel</surname><given-names>E</given-names> </name></person-group><article-title>The effect of perceived burdensomeness and thwarted belongingness on therapists&#x2019; assessment of patients&#x2019; suicide risk</article-title><source>Psychother Res</source><year>2016</year><month>07</month><volume>26</volume><issue>4</issue><fpage>436</fpage><lpage>445</lpage><pub-id pub-id-type="doi">10.1080/10503307.2015.1013161</pub-id><pub-id pub-id-type="medline">25751580</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>So</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Aligning large language models for enhancing psychiatric interviews through symptom delineation and summarization: pilot study</article-title><source>JMIR Form Res</source><year>2024</year><month>10</month><day>24</day><volume>8</volume><fpage>e58418</fpage><pub-id pub-id-type="doi">10.2196/58418</pub-id><pub-id pub-id-type="medline">39447159</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>R</given-names> </name></person-group><article-title>Integrative diagnosis of psychiatric conditions using ChatGPT and fMRI data</article-title><source>BMC Psychiatry</source><year>2025</year><month>02</month><day>19</day><volume>25</volume><issue>1</issue><fpage>145</fpage><pub-id pub-id-type="doi">10.1186/s12888-025-06586-w</pub-id><pub-id pub-id-type="medline">39972267</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chanteclair</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lartigau</surname><given-names>M</given-names> </name><name name-style="western"><surname>Salles</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Assessing the acceptability of a sleep-targeted digital intervention among geriatric inpatients: a preliminary study</article-title><source>Digit Health</source><year>2025</year><month>01</month><volume>11</volume><pub-id pub-id-type="doi">10.1177/20552076241293935</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>von Eschenbach</surname><given-names>WJ</given-names> </name></person-group><article-title>Transparency and the black box problem: why we do not trust AI</article-title><source>Philos Technol</source><year>2021</year><month>12</month><volume>34</volume><issue>4</issue><fpage>1607</fpage><lpage>1622</lpage><pub-id pub-id-type="doi">10.1007/s13347-021-00477-0</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Search equations for PubMed and other databases.</p><media xlink:href="mental_v13i1e93672_app1.docx" xlink:title="DOCX File, 142 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Risk of bias assessment.</p><media xlink:href="mental_v13i1e93672_app2.docx" xlink:title="DOCX File, 1181 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 1</label><p>PRISMA checklist.</p><media xlink:href="mental_v13i1e93672_app3.docx" xlink:title="DOCX File, 3311 KB"/></supplementary-material></app-group></back></article>