<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Ment Health</journal-id><journal-id journal-id-type="publisher-id">mental</journal-id><journal-id journal-id-type="index">16</journal-id><journal-title>JMIR Mental Health</journal-title><abbrev-journal-title>JMIR Ment Health</abbrev-journal-title><issn pub-type="epub">2368-7959</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v13i1e99185</article-id><article-id pub-id-type="doi">10.2196/99185</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Interpretable Topic Modeling of Spontaneous Speech in Depression Using Large Language Models: Multilingual Four-Cohort Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Cortal</surname><given-names>Gustave</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Guessoum</surname><given-names>S&#x00E9;lim Benjamin</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cao</surname><given-names>Xuan-Nga</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>de Leon-Martinez</surname><given-names>Santiago</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Baca-Garc&#x00ED;a</surname><given-names>Enrique</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="aff" rid="aff9">9</xref><xref ref-type="aff" rid="aff10">10</xref><xref ref-type="aff" rid="aff11">11</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Riad</surname><given-names>Rachid</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Callyope</institution><addr-line>P&#x00E9;pini&#x00E8;re d'entreprises Paris Sant&#x00E9; Cochin, 29 Rue du Faubourg Saint-Jacques</addr-line><addr-line>Paris</addr-line><country>France</country></aff><aff id="aff2"><institution>Universit&#x00E9; Paris-Saclay, Centre de recherche en Epid&#x00E9;miologie et Sant&#x00E9; des Populations</institution><addr-line>Villejuif</addr-line><country>France</country></aff><aff id="aff3"><institution>Brno University of Technology</institution><addr-line>Brno</addr-line><country>Czech Republic</country></aff><aff id="aff4"><institution>Kempelen Institute of Intelligent Technologies</institution><addr-line>Bratislava</addr-line><country>Slovakia</country></aff><aff id="aff5"><institution>Department of Psychiatry, Hospital Universitario Fundaci&#x00F3;n Jim&#x00E9;nez D&#x00ED;az</institution><addr-line>Madrid</addr-line><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff6"><institution>Department of Psychiatry, University Hospital Rey Juan Carlos</institution><addr-line>Mostoles</addr-line><country>Spain</country></aff><aff id="aff7"><institution>Department of Psychiatry, Hospital Universitario General de Villalba</institution><addr-line>Madrid</addr-line><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff8"><institution>Department of Psychiatry, University Hospital Infanta Elena</institution><addr-line>Valdemoro</addr-line><country>Spain</country></aff><aff id="aff9"><institution>Department of Psychology, Catholic University of the Maule</institution><addr-line>Talca</addr-line><addr-line>Maule Region</addr-line><country>Chile</country></aff><aff id="aff10"><institution>Department of Psychiatry, Universidad Aut&#x00F3;noma de Madrid</institution><addr-line>Madrid</addr-line><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff11"><institution>Centro de Investigaci&#x00F3;n Biom&#x00E9;dica en Red de Salud Mental</institution><addr-line>Madrid</addr-line><addr-line>Madrid</addr-line><country>Spain</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Torous</surname><given-names>John</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Gopinath</surname><given-names>Avinash</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Lau</surname><given-names>Gabriel Rongyang</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Podder</surname><given-names>Soumyajit</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Rachid Riad, PhD, Callyope, P&#x00E9;pini&#x00E8;re d'entreprises Paris Sant&#x00E9; Cochin, 29 Rue du Faubourg Saint-Jacques, Paris, 75014, France, 33 0666522141; <email>rachid@callyope.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>17</day><month>8</month><year>2026</year></pub-date><volume>13</volume><elocation-id>e99185</elocation-id><history><date date-type="received"><day>23</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>20</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>21</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Gustave Cortal, S&#x00E9;lim Benjamin Guessoum, Xuan-Nga Cao, Santiago de Leon-Martinez, Enrique Baca-Garc&#x00ED;a, Rachid Riad. Originally published in JMIR Mental Health (<ext-link ext-link-type="uri" xlink:href="https://mental.jmir.org">https://mental.jmir.org</ext-link>), 17.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Mental Health, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mental.jmir.org/">https://mental.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mental.jmir.org/2026/1/e99185"/><abstract><sec><title>Background</title><p>Depression is underdiagnosed worldwide, and clinicians rely on interpreting patients&#x2019; subjective speech. Qualitative analysis of patient language does not scale, and existing computational approaches describe topics with keyword lists that miss clinical nuance.</p></sec><sec><title>Objective</title><p>We evaluated whether clustering spontaneous speech transcripts with large language models (LLMs) yields clusters whose membership is associated with validated clinical scales across multilingual cohorts, and whether LLMs can render those clusters human-readable through fine-grained natural-language descriptions. We further examined which interview questions yield clusters most strongly associated with clinical status, and how sociodemographic factors relate to cluster membership.</p></sec><sec sec-type="methods"><title>Methods</title><p>We analyzed spontaneous speech transcripts from 4 independent cohorts: a French general population sample (1338 participants) and 3 clinical samples in Italian (n=116), Chinese (n=52), and Spanish (n=90). Responses to open-ended questions were transcribed, embedded with a multilingual language model, dimensionally reduced, and grouped by density-based clustering. An LLM then summarized each cluster into a natural-language description. Cluster membership was tested for association with validated clinical scales (Patient Health Questionnaire-9, Beck Depression Inventory, Generalized Anxiety Disorder 7-item scale, Athens Insomnia Scale, Multidimensional Fatigue Inventory, and Columbia Suicide Severity Rating Scale), clinician-assigned depression diagnoses, and sociodemographic factors (age, education, and sex).</p></sec><sec sec-type="results"><title>Results</title><p>Unsupervised clustering yielded clusters significantly associated with clinical scores in the French, Italian, and Chinese cohorts, with an exploratory association in the smaller Spanish cohort. In the French general population, Patient Health Questionnaire-9 depression scores differed across clusters (&#x03B7;&#x00B2;=0.19, 95% CI 0.17 to 0.24, <italic>P</italic>&#x003C;.001), as did anxiety (Generalized Anxiety Disorder 7-item scale), insomnia (Athens Insomnia Scale), and fatigue (Multidimensional Fatigue Inventory) scores (&#x03B7;&#x00B2;=0.14 to 0.16, all <italic>P</italic>&#x003C;.001). In the clinical cohorts, cluster membership was associated with clinician-diagnosed depression in the Italian sample (Cram&#x00E9;r <italic>V</italic>=0.74, 0.65 to 0.85, <italic>P</italic>&#x003C;.001) and with major depressive disorder diagnosis in the Chinese sample (<italic>V</italic>=0.55, 0.36 to 0.78, <italic>P</italic>=.001). A suicide-risk association in the smaller Spanish sample (Columbia Suicide Severity Rating Scale) was unstable and is reported as exploratory (mean <italic>V</italic>=0.27). The feelings-and-sleep question yielded the strongest clinical associations, whereas questions about past or future events yielded small effect sizes (&#x03B7;&#x00B2;&#x2264;0.05). Age (&#x03B7;&#x00B2;=0.27, 0.24 to 0.32) and sex (Cram&#x00E9;r <italic>V</italic>=0.22, 0.21 to 0.30) were each associated with cluster membership for the &#x201C;describe your last 24 hours&#x201D; question (both <italic>P</italic>&#x003C;.001).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Fine-grained LLM-based topic modeling automates the labor-intensive early stages of thematic analysis, in minutes of compute, and recovers core psychiatric constructs across languages. It produces human-readable cluster descriptions. Certain interview questions yield clusters more strongly associated with clinical status than others, and sociodemographic factors shape topic content independently of clinical status, an often-overlooked confound. As the cohorts differ, cross-cohort observations are preliminary without generalizability. Screening, treatment-planning, and decision-support uses remain to be established prospectively.</p></sec></abstract><kwd-group><kwd>depression</kwd><kwd>thematic analysis</kwd><kwd>speech analysis</kwd><kwd>large language models</kwd><kwd>topic modeling</kwd><kwd>natural language processing</kwd><kwd>mental health</kwd><kwd>computational psychiatry</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Depressive disorders are among the major challenges in psychiatry. These highly prevalent conditions are estimated to affect 30% to 40% of individuals during their life, across genders and regions worldwide [<xref ref-type="bibr" rid="ref1">1</xref>]. Depression is a leading contributor to global disability and affects functioning and mortality. Depression remains underdiagnosed and undertreated worldwide [<xref ref-type="bibr" rid="ref2">2</xref>].</p><p>Psychiatrists face distinct, complex assessment challenges in comparison to other medical specialties. Most medical fields rely on complementary investigations and paraclinical tests to assess the objective status of patients. Psychiatrists&#x2019; evaluation depends on interpreting patients&#x2019; subjective experiences in their own words, then mapping these diverse narratives onto standardized <italic>DSM</italic> (<italic>Diagnostic and Statistical Manual of Mental Disorders</italic>) criteria [<xref ref-type="bibr" rid="ref3">3</xref>]. This &#x201C;transformation&#x201D; task, from individual expressions of suffering to fixed sets of symptoms and diagnoses, yields variable and modest interrater reliability [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. Moreover, current standardized tools in psychiatry consist of rating scales (self-reported and clinician-administered) that rely on pre-established constructs [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Even when useful, they do not capture the full complexity of the lived experience of affective disorders [<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>This modest interrater reliability is not due to inadequate clinical practice, but rather the inherent complexity of psychiatric phenomena and the current limitations of validated assessment tools. In 2021, while cardiology has over 700 US Food and Drug Administration&#x2013;approved devices and neurology over 300 devices to support clinical decision-making, psychiatry has fewer than 20 devices [<xref ref-type="bibr" rid="ref9">9</xref>], and no AI-powered devices have been adopted despite AI&#x2019;s integration across other specialties [<xref ref-type="bibr" rid="ref10">10</xref>]. This technology gap is particularly surprising given that patient language, the primary data source in psychiatric consultations, is rarely analyzed in a systematic way despite extensive evidence linking linguistic patterns to mood disorder symptomatology [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>Language analysis and interpretation, or the qualitative analysis of patients&#x2019; speech content, is a central part of clinical practice. It helps clinicians understand patients&#x2019; narratives about themselves, their lives, and their illnesses, and it forms the basis of psychotherapy. However, producing reliable data on this topic is highly resource-intensive, as shown in a prior study [<xref ref-type="bibr" rid="ref14">14</xref>]. This study required extensive collaborative workshops and global expert panels to identify how depression manifests through specific linguistic themes such as altered temporal experience, embodied distress, and existential emptiness. There is also a prognostic value when these themes are extracted: how patients narrate autobiographical memories predicts depression trajectory, with overgeneral vs specific memory styles serving as age-dependent markers [<xref ref-type="bibr" rid="ref15">15</xref>]. Natural language processing has been used to analyze such narrative features automatically, classifying emotions in personal narratives from their underlying psychological components [<xref ref-type="bibr" rid="ref16">16</xref>] and relating stylistic patterns to psychological states [<xref ref-type="bibr" rid="ref17">17</xref>]. Yet, clinicians struggle to analyze and interpret the narratives of patients from minority groups, particularly when cultural differences shape expressions of distress [<xref ref-type="bibr" rid="ref18">18</xref>]. While language analysis is routine in clinical psychiatry, producing reliable scientific data, generating fine-grained annotations, and conducting such analyses on large populations remain challenging.</p><p>Moreover, thematic analysis is particularly relevant in psychiatry because it resonates with the clinical focus on how people construct meaning from their experiences. As a qualitative method, thematic analysis identifies recurrent patterns in textual data and offers a structured approach to interpreting narratives through iterative annotation and inductive refinement of higher-order themes [<xref ref-type="bibr" rid="ref19">19</xref>]. It can reveal phenomenological nuances during psychiatric disorders: how individuals experience emotions, self, embodiment, and social relatedness. The method has been used across diverse populations and data sources. However, when conducted manually, it is labor-intensive, often requiring trained coders or panels of expert clinicians to produce clinically validated themes [<xref ref-type="bibr" rid="ref14">14</xref>]. It is also sensitive to clinician bias and typically constrained to small, monolingual corpora, limiting its scalability and generalizability [<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>Computational approaches to thematic analysis (commonly named topic modeling in natural language processing) represent documents in a shared vector space and discover structure via clustering or probabilistic topic models. Canonical methods (eg, latent Dirichlet allocation) treat documents as mixtures of latent topics [<xref ref-type="bibr" rid="ref20">20</xref>]. Neural variants such as BERTopic embed textual data with transformer-based language models, reduce dimensionality, cluster, and summarize each cluster with keywords [<xref ref-type="bibr" rid="ref21">21</xref>]. More recently, prompt-based frameworks delegate labeling to large language models (LLMs), yielding more specific, interpretable topic descriptions [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. In mental health, topic modeling has been applied to peer-support forums [<xref ref-type="bibr" rid="ref24">24</xref>], pandemic-era shifts in Reddit help-seeking [<xref ref-type="bibr" rid="ref25">25</xref>], depression forums [<xref ref-type="bibr" rid="ref26">26</xref>], and question-and-answer platforms [<xref ref-type="bibr" rid="ref27">27</xref>]. Most relevant to our setting, it has been applied to free-response clinical speech, identifying depression-related topics in smartphone-collected recordings transcribed by automatic speech recognition [<xref ref-type="bibr" rid="ref28">28</xref>].</p><p>Computational approaches offer 3 gains: reproducible annotation that limits researcher bias, substantial time savings over labor-intensive manual annotation, and the capacity to analyze larger datasets. However, many automated analyses remain shallow, relying on surface-level lexical statistics (eg, keyword counts and part-of-speech frequencies) to describe clusters [<xref ref-type="bibr" rid="ref29">29</xref>], which can miss context, pragmatics (eg, sarcasm), and cultural idioms. Prior studies have also rarely addressed potential confounding factors. Language use may vary with demographics such as age and gender, independently of mental health status [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>]. For example, adolescents with depression tend to discuss friends and school, whereas adults with depression might focus more on work and financial stressors [<xref ref-type="bibr" rid="ref33">33</xref>]. Most topic modeling studies for mental health focus on a single language and data source (eg, one social media or one country&#x2019;s clinical sample) [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref36">36</xref>]. Findings from a single cohort may not translate to other populations [<xref ref-type="bibr" rid="ref37">37</xref>]. As a result, it remains unclear whether the topics identified in one group (eg, English-speaking social media users) also appear in other cultural or linguistic groups.</p><p>Fine-grained descriptions of clusters (eg, examination-related rumination and job-loss-related hopelessness) capture clinically relevant distinctions that broad, keyword-based descriptions (eg, sleep, work, and family) collapse together. We generated cluster descriptions using language models as they provide natural-language, human-readable labels, overcoming the interpretability limits of keyword-based topic models while retaining scalability for large corpora.</p><p>We developed a multilingual pipeline that (1) clusters spontaneous speech transcripts from 4 cohorts (general population and 3 clinical samples in French, Italian, Chinese, and Spanish), (2) generates fine-grained natural-language descriptions for each cluster, and (3) tests cluster membership for association with clinical scores and sociodemographic factors. We then quantified which questions best separate clinically relevant clusters, and we measured confounding by age, education, and sex, an aspect often overlooked in topic modeling despite strong demographic effects on language use.</p><p>Across these cohorts, cluster membership was associated with validated clinical scales, robustly in 3 cohorts and as an exploratory finding in the smaller Spanish cohort, and the language-model descriptions render the clusters readable.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>In this study, we automatically clustered mental health transcripts from 4 distinct cohorts. Then, we generated descriptions of clusters to find fine-grained topics. Finally, we performed statistical analysis to find clinical differences between the clusters. Our automatic pipeline used language models for both clustering transcripts and generating cluster descriptions.</p></sec><sec id="s2-2"><title>Study Design</title><p>This is a secondary cross-sectional analysis of spontaneous speech transcripts from 4 independent observational cohorts: a large French general population sample (1809 assessments from 1338 participants) and 3 clinical populations collected in Italian (n=116), Chinese (n=52), and Spanish (n=90) languages. The data comprised transcription of participants&#x2019; spontaneous speech productions in response to open-ended questions and validated self- or clinician-administered psychiatric rating scales. In <xref ref-type="table" rid="table1">Table 1</xref>, we summarized statistics of demographics and clinical scores of the 4 cohorts. Categorical variables were compared with the Pearson chi-square test, and continuous variables were compared with the Kruskal-Wallis H test based on control and noncontrol groups.</p><p>Throughout, the unit of analysis for clustering is the individual transcript (1 participant may contribute several transcripts, 1 per interview question). In <xref ref-type="table" rid="table1">Table 1</xref>, the 3 clinical cohorts are summarized at the participant level (Italian n=116, Chinese n=52, and Spanish n=90), whereas the French general population is summarized at the assessment level (1809 assessments from 1338 participants), reflecting how each cohort was collected. The French demographic and clinical counts in the table are therefore per assessment. The number of transcripts analyzed per cohort and question, after removing duplicates and transcripts shorter than 20 characters, is reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, which also reports the participant, recording, and transcript counts for each cohort.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Demographics and clinical scores of the 4 cohorts. Significance indicators test whether each variable differs between clinical (depressed/suicidal) and control groups within each cohort: <italic>P</italic>&#x003C;.001 indicates a significant difference. For clinical scores (PHQ-9<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>), <italic>P</italic>&#x003C;.001 reflects the expected clinical difference between groups. In the Chinese cohort, a PHQ-9 score above 10 was part of the patient inclusion criteria, so this difference partly reflects that inclusion criterion. Depression diagnosis is a clinician-assigned clinical diagnosis (see Methods for the criteria used in each cohort). A variable that defines a cohort&#x2019;s clinical grouping, the PHQ-9 in the French cohort, the depression diagnosis in the Italian and Chinese cohorts, and the C-SSRS<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> class in the Spanish cohort, is not tested against that grouping. Tests: Pearson chi-square (categorical) and Kruskal-Wallis H (continuous).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">General population (n=1809 assessments)</td><td align="left" valign="bottom">Androids (n=116)</td><td align="left" valign="bottom">MODMA<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> (n=52)</td><td align="left" valign="bottom">VOCES<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> (n=90)</td></tr></thead><tbody><tr><td align="left" valign="top">Demographics</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Language</td><td align="left" valign="top">French</td><td align="left" valign="top">Italian</td><td align="left" valign="top">Chinese</td><td align="left" valign="top">Spanish</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age</td><td align="left" valign="top"><italic>P</italic>&#x003C;.001</td><td align="left" valign="top">n.s.<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top">n.s.</td><td align="left" valign="top"><italic>P</italic>&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mean (SD)</td><td align="left" valign="top">37.8 (18.2)</td><td align="left" valign="top">37.4 (12.0)</td><td align="left" valign="top">31.3 (9.2)</td><td align="left" valign="top">38.6 (14.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Range</td><td align="left" valign="top">18&#x2010;91</td><td align="left" valign="top">19&#x2010;71</td><td align="left" valign="top">18&#x2010;52</td><td align="left" valign="top">21&#x2010;76</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sex, n (<bold>%</bold>)</td><td align="left" valign="top">n.s.</td><td align="left" valign="top">n.s.</td><td align="left" valign="top">n.s.</td><td align="left" valign="top">n.s.</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">1187 (66.2)</td><td align="left" valign="top">84 (72.4)</td><td align="left" valign="top">16 (31)</td><td align="left" valign="top">39 (43)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">595 (33.2)</td><td align="left" valign="top">32 (27.6)</td><td align="left" valign="top">36 (69)</td><td align="left" valign="top">48 (53)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other</td><td align="left" valign="top">11 (0.6)</td><td align="left" valign="top">0 (0.0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">3 (3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Education, n (<bold>%</bold>)</td><td align="left" valign="top">n.s.</td><td align="left" valign="top">n.s.</td><td align="left" valign="top">n.s.</td><td align="left" valign="top">N/A<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No diploma</td><td align="left" valign="top">52 (2.9)</td><td align="left" valign="top">11 (9.5)</td><td align="left" valign="top">7 (13)</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Secondary</td><td align="left" valign="top">291 (16.2)</td><td align="left" valign="top">37 (31.9)</td><td align="left" valign="top">8 (15)</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Higher short</td><td align="left" valign="top">213 (11.9)</td><td align="left" valign="top">52 (44.8)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Higher long</td><td align="left" valign="top">1236 (69.0)</td><td align="left" valign="top">16 (13.8)</td><td align="left" valign="top">37 (71)</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top">Clinical evaluation</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>C-SSRS</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Suicidal risk, n (%)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">60 (67)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No suicidal risk, n (%)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">30 (33)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Depression diagnosis</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Depression, n (%)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">64 (55.2)</td><td align="left" valign="top">23 (44)</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No depression, n (%)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">52 (44.8)</td><td align="left" valign="top">29 (56)</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>PHQ-9</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top"><italic>P</italic>&#x003C;.001</td><td align="left" valign="top"><italic>P</italic>&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mean (SD)</td><td align="left" valign="top">5.2 (4.6)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">9.4 (8.5)</td><td align="left" valign="top">10.5 (6.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Range</td><td align="left" valign="top">0&#x2010;27</td><td align="left" valign="top">N/A</td><td align="left" valign="top">0&#x2010;25</td><td align="left" valign="top">0&#x2010;26</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2265;10, n (%)</td><td align="left" valign="top">290 (16.0)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>PHQ-9: Patient Health Questionnaire-9.</p></fn><fn id="table1fn2"><p><sup>b</sup>C-SSRS: Columbia Suicide Severity Rating Scale.</p></fn><fn id="table1fn3"><p><sup>c</sup>MODMA: multi-modal open dataset for mental-disorder analysis.</p></fn><fn id="table1fn4"><p><sup>d</sup>VOCES: Spanish clinical cohort for suicide-risk research.</p></fn><fn id="table1fn5"><p><sup>e</sup>n.s.: indicates <italic>P</italic>&#x2265;.05 (no significant difference between groups).</p></fn><fn id="table1fn6"><p><sup>f</sup>N/A: not applicable.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-3"><title>Recruitment</title><sec id="s2-3-1"><title>French General Population Cohort</title><p>The French general population cohort contained 1809 assessment visits from 1338 French-speaking participants (1161 contributed a single visit and the remainder 2 to 6 visits, with 1 participant assessed 16 times). The unit of analysis is the per-question transcript. As some participants contributed repeated visits, a one-visit-per-participant sensitivity analysis (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>) leaves the associations between cluster membership and clinical scores essentially unchanged. They responded to open-ended questions, including the following: &#x201C;describe your last 24 hours,&#x201D; &#x201C;describe a negative event that happened to you in the past,&#x201D; &#x201C;describe a positive event that happened to you in the past,&#x201D; &#x201C;describe a negative event or situation you think might happen in the future (wk, mo, or y),&#x201D; &#x201C;describe a positive event or situation you think might happen in the future (wk, mo, or y),&#x201D; and &#x201C;describe how you are feeling at the moment and how your sleep has been lately.&#x201D;</p><p>Self-administered questionnaires assessed participants&#x2019; mental health status, including depressive symptoms using the Patient Health Questionnaire-9 (PHQ-9) [<xref ref-type="bibr" rid="ref38">38</xref>] and the Beck Depression Inventory (BDI) [<xref ref-type="bibr" rid="ref39">39</xref>]; anxiety using the Generalized Anxiety Disorder 7-item scale (GAD-7); insomnia using the Athens Insomnia Scale (AIS) [<xref ref-type="bibr" rid="ref40">40</xref>]; and fatigue using the Multidimensional Fatigue Inventory (MFI) [<xref ref-type="bibr" rid="ref41">41</xref>]. Using the PHQ-9 score, we obtained a control group (PHQ-9 &#x2264;9) and a probable-depression group (PHQ-9 &#x2265;10). PHQ-9 was at or above 10, the conventional threshold for probable depression, in 290 of 1809 (16%; 221 of 1338, 16.5% participants at first visit) assessments.</p><p>Participants completed a series of speech tasks and self-administered questionnaires through a mobile research app specifically designed for clinical studies.</p></sec><sec id="s2-3-2"><title>Italian Clinical Population (Androids) Cohort</title><p>The Androids cohort was specifically designed for automatic depression detection from speech [<xref ref-type="bibr" rid="ref42">42</xref>], and includes 228 recordings from 116 native Italian speakers, collected across 2 tasks (a spontaneous interview task and a read-speech task). We analyzed the spontaneous interview task only. After removing duplicates and transcripts shorter than 20 characters, 110 transcripts entered the clustering analysis (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Of the 116 participants, 64 participants were clinically diagnosed with depression by professional psychiatrists, providing a reliable characterization of individuals with depression compared to self-reported assessments. The interview task involved answering questions about everyday life (eg, &#x201C;What did you do last weekend?&#x201D;).</p><p>Participants were native Italian speakers recruited across 5 Mental Health Centers in Italy. Depression was diagnosed by psychiatrists according to <italic>DSM-5</italic> (<italic>Diagnostic and Statistical Manual of Mental Disorders</italic> [Fifth Edition]) criteria, spanning major depressive disorder (MDD), bipolar disorder in a depressive phase, and related depressive conditions [<xref ref-type="bibr" rid="ref42">42</xref>]. The Androids corpus provides this binary clinical diagnosis (depressed vs control). Controls reported no prior mental health history and were selected to match the patient group on age, gender, and education.</p></sec><sec id="s2-3-3"><title>Chinese Clinical Population (MODMA) Cohort</title><p>The multi-modal open dataset for mental-disorder analysis (MODMA) cohort [<xref ref-type="bibr" rid="ref43">43</xref>] included 52 Chinese-speaking participants, with 23 outpatients diagnosed with MDD (16 males and 7 females, aged 18&#x2010;52 years) and 29 healthy controls (20 males and 9 females, aged 19&#x2010;52 years). MDD diagnoses were confirmed by clinical psychiatrists at Lanzhou University Second Hospital based on the Mini-International Neuropsychiatric Interview [<xref ref-type="bibr" rid="ref44">44</xref>] and <italic>DSM-IV</italic> (<italic>Diagnostic and Statistical Manual of Mental Disorders</italic> [Fourth Edition]) diagnostic criteria [<xref ref-type="bibr" rid="ref45">45</xref>].</p><p>Speech samples consisted of responses to 18 interview questions derived from <italic>DSM-IV</italic> and the Hamilton Depression Rating Scale [<xref ref-type="bibr" rid="ref46">46</xref>]. Following the dataset design [<xref ref-type="bibr" rid="ref43">43</xref>], these questions are grouped by emotional valence into 6 positive, 6 neutral, and 6 negative questions. The analysis reported here and shown in <xref ref-type="fig" rid="figure1">Figure 1</xref> uses the neutral-valence questions, everyday topics such as &#x201C;describe one of your friends, including age, job, character, and hobbies.&#x201D; For each participant, the answers to the questions within a valence group were concatenated into a single transcript before clustering. The positive-valence questions (such as &#x201C;What is the best gift you have ever received, and how did you feel?&#x201D;) and the negative-valence questions (such as &#x201C;What makes you desperate?&#x201D;) were analyzed separately (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Proportion of participants with depression across automatically identified clusters in the MODMA cohort (n=52 transcripts, Chinese clinical population). Responses to the neutral-valence interview questions (each participant&#x2019;s answers to that valence group concatenated; see Methods section). Depression status is a clinician-assigned MDD diagnosis (MINI/<italic>DSM-IV</italic> criteria). Effect size Cram&#x00E9;r V=0.55 (95% CI 0.36 to 0.78, <italic>P</italic>=.001). Cluster descriptions generated by Qwen3-14B summarizing 30 random transcripts per cluster. In cluster 1 (age 26&#x00B1;6 years, n=10), the individuals express a range of emotional states from mild anxiety and stress related to academic or work pressures to generally positive moods, emphasize the significance of maintaining healthy family relationships and having supportive friends or mentors, acknowledge personal traits such as being easily irritable, argumentative, or overly accommodating, and outline specific plans involving career development, returning to hometowns, or improving personal skills such as research capabilities or education. In cluster 4 (age 28&#x00B1;6 years, n=14), the individuals express how physical health issues and emotional struggles such as depression, anxiety, and loss of motivation significantly impact their daily lives, relationships, and work, while also highlighting the complex role of supportive friends and family, as well as persistent feelings of inadequacy and communication challenges within familial contexts. <italic>DSM-IV</italic>: <italic>Diagnostic and Statistical Manual of Mental Disorders</italic> (Fourth Edition); MDD: major depressive disorder; MINI: Mini-International Neuropsychiatric Interview; MODMA: multimodal open dataset for mental-disorder analysis.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e99185_fig01.png"/></fig><p>Inclusion criteria included a PHQ-9 score above 10 and no psychotropic drug treatment in the 2 weeks before participation. Healthy controls were recruited through public advertisements, with exclusion criteria that ruled out personal or family histories of mental disorders.</p></sec><sec id="s2-3-4"><title>Spanish Clinical Population (VOCES) Cohort</title><p>The Spanish clinical cohort for suicide-risk research (VOCES) cohort [<xref ref-type="bibr" rid="ref47">47</xref>] comprised 90 Spanish-speaking adults who responded to open-ended questions such as: &#x201C;Can you recall some recent good news you had and how did that make you feel?&#x201D; &#x201C;Do you get a characteristic feeling when you&#x2019;re sad or down, and what do you normally do to cheer yourself up?&#x201D; and &#x201C;Describe your average day.&#x201D;</p><p>Suicidal risk was assessed by professional psychiatrists using the validated Spanish version of the Columbia Suicide Severity Rating Scale (C-SSRS) [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref49">49</xref>]. Each participant&#x2019;s summed C-SSRS severity score ranged from 0 to 21 in this cohort (mean 9.2, SD 7.5). For analysis, participants were dichotomized into a suicidal-risk group, defined as any positive C-SSRS finding (a score of 1 or more, indicating suicidal ideation or behavior), and a no-risk group (a score of 0). This binary risk class is the outcome tested in <xref ref-type="fig" rid="figure2">Figure 2</xref>. A total of 60 of the 90 (67%) participants were in the suicidal-risk group.</p><p>Inclusion criteria required participants to be at least 18 years of age, fluent in Spanish, and capable of understanding and signing the informed-consent form. Psychiatric outpatients were recruited consecutively during routine consultations, and healthy volunteers were enrolled through snowball sampling among mental-health staff and university students.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Proportion of participants with suicidal risk across automatically identified clusters in the VOCES cohort (n=58 transcripts, Spanish clinical population). Responses to &#x201C;do you get a characteristic feeling when you&#x2019;re sad or down, and what do you normally do to cheer yourself up?&#x201D; Suicidal-risk class (any positive C-SSRS finding vs none), assessed by psychiatrists with the Spanish C-SSRS. Due to the small sample, this association is unstable across clustering seeds, so it is treated as exploratory and its effect size is the seed-averaged value (mean Cram&#x00E9;r V=0.27; <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). The figure shows the primary clustering, the single run on which the cluster descriptions and the plotted proportions are based, and for this run Cram&#x00E9;r V=0.50 (95% CI 0.37 to 0.71, <italic>P</italic>=.006). Cluster descriptions generated by Qwen3-14B summarizing 30 random transcripts per cluster. In cluster 1 (age 32&#x00B1;13 years, n=13), the individuals express common strategies for coping with sadness or low mood, including mindfulness practices (meditation or breathing exercises), seeking social support (talking to friends or loved ones), engaging in physical activity or rest (sleep or exercise), and allowing emotions to pass naturally without forced suppression or avoidance. In cluster 5 (age 34&#x00B1;11 years, n=9), the individuals express common strategies for coping with low moods, including using media (eg, television, YouTube, and music), engaging in physical activities (eg, walking or cleaning), seeking social support (eg, talking to friends or family), and using distraction techniques (eg, gaming or napping) to avoid rumination, while some emphasize the importance of acceptance, structured routines, and medication in managing emotional distress. C-SSRS: Columbia Suicide Severity Rating Scale; VOCES: Spanish clinical cohort for suicide-risk research.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e99185_fig02.png"/></fig></sec></sec><sec id="s2-4"><title>Analysis</title><sec id="s2-4-1"><title>Overview</title><p>We adopted a data-driven pipeline that clusters transcripts based on their semantic similarity using multilingual language models (<xref ref-type="fig" rid="figure3">Figure 3</xref>). No machine translation was performed as transcripts were analyzed in the original language. Hyperparameters for each step are available in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Computational pipeline for semantic clustering and description generation. Speech transcripts are converted to semantic vectors via multilingual embeddings, dimensionally reduced, and grouped into clusters using density-based methods. Each cluster undergoes: (1) statistical analysis linking cluster membership to clinical and demographic variables, and (2) automatic description generation where a large language model summarizes transcripts per cluster into human-readable natural language. This dual quantitative-qualitative output enables both statistical hypothesis testing and human-readable interpretation of discovered topics (<xref ref-type="fig" rid="figure1">Figures 1</xref>, <xref ref-type="fig" rid="figure2">2</xref>, <xref ref-type="fig" rid="figure4">4</xref>, and <xref ref-type="fig" rid="figure5">5</xref>).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e99185_fig03.png"/></fig><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Distribution of PHQ-9 depression scores across automatically identified clusters in the French general population cohort (n=1645 transcripts in 26 clusters, after excluding HDBSCAN noise). Responses to the prompt &#x201C;describe how you are feeling at the moment and how your sleep has been lately.&#x201D; Box plots show median (center line), IQR (box), and 1.5&#x00D7; IQR whiskers. Effect size &#x03B7;&#x00B2;=0.19 (95% CI 0.17 to 0.24, <italic>P</italic>&#x003C;.001) indicates a large difference between clusters per Cohen conventions. Sample sizes per cluster range from 34 to 92. Selected cluster descriptions (clusters 1, 10, 12, and 26) were generated by Qwen3-14B summarizing 30 random transcripts per cluster. In cluster 1 (age 39&#x00B1;19 years, n=92), the individuals express consistent satisfaction with their current well-being, emphasizing good sleep quality, restful or pleasant nights, and a general sense of relaxation, even when noting variations in sleep duration or occasional fatigue. In cluster 10 (age 69&#x00B1;15 years, n=34), the individuals express frequent nighttime urinary interruptions disrupting sleep, often attributed to age-related conditions such as prostate issues or overactive bladder, alongside mixed reports of physical well-being, mental resilience, and lifestyle factors such as retirement or exercise influencing their overall health and sleep patterns. In cluster 12 (age 24&#x00B1;9 years, n=67), the individuals express stress related to academic examinations, significant life decisions, and workloads, alongside sleep disturbances caused by lifestyle changes, increased responsibilities, or environmental adjustments, while some also highlight temporary relief from pressures through personal achievements or upcoming positive events. Cluster 26 (age 25&#x00B1;9 years, n=37), the individuals express sleep disturbances characterized by insomnia, frequent awakenings, and restless sleep, alongside pervasive anxiety, emotional instability, and self-esteem issues, which collectively contribute to persistent fatigue, impaired daily functioning, and a diminished sense of well-being. HDBSCAN: Hierarchical Density-Based Spatial Clustering of Applications With Noise; PHQ-9: Patient Health Questionnaire-9.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e99185_fig04.png"/></fig><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Proportion of participants with depression across automatically identified clusters in the Androids cohort (n=106 transcripts, Italian clinical population, after excluding HDBSCAN noise). Responses to questions about everyday life (eg, &#x201C;What did you do last weekend?&#x201D;). Depression status is a clinician-assigned <italic>DSM-5</italic> diagnosis. Effect size Cram&#x00E9;r V=0.74 (95% CI 0.65 to 0.85, <italic>P</italic>&#x003C;.001). Cluster descriptions generated by Qwen3-14B summarizing 30 random transcripts per cluster. In cluster 1 (age 44&#x00B1;11 years, n=19), the individuals express a shared focus on family dynamics and intergenerational relationships, regional identity and emotional ties to Naples, and the pursuit of personal passions such as music, research, and education, while reflecting on the complexities of balancing domestic responsibilities with professional or creative endeavors. In cluster 5 (age 50&#x00B1;11 years, n=30), the individuals express persistent struggles with depression, anxiety, and panic attacks, often exacerbated by significant life events such as family responsibilities, health-related stress from routine medical check-ups, and a pervasive sense of stagnation and purposelessness amid repetitive daily routines, while some also grapple with disrupted academic or career goals and complex family dynamics. <italic>DSM-5</italic>: <italic>Diagnostic and Statistical Manual of Mental Disorders</italic> (Fifth Edition); HDBSCAN: Hierarchical Density-Based Spatial Clustering of Applications With Noise.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mental_v13i1e99185_fig05.png"/></fig></sec><sec id="s2-4-2"><title>Audio Transcription</title><p>We transcribed all speech recordings with Whisper large-v3-turbo, a state-of-the-art neural model for automatic speech recognition [<xref ref-type="bibr" rid="ref50">50</xref>]. We removed duplicates and small transcripts (less than 20 characters).</p></sec><sec id="s2-4-3"><title>Text Embedding</title><p>To obtain multilingual semantic representations, each transcript was mapped to a 4096-dimensional vector (also called an embedding) using Qwen3-Embedding-8B, a language model for text embedding [<xref ref-type="bibr" rid="ref51">51</xref>]. We selected this model because, as of June 2025, it ranked first on the multilingual track of the Massive Text Embedding Benchmark [<xref ref-type="bibr" rid="ref52">52</xref>], covering all 4 languages of our cohorts, and because it is released under a permissive open-weight (Apache-2.0) license, which allowed us to run the model ourselves, a requirement for processing sensitive clinical speech without transmitting data to any third-party or commercial service. The model turned every transcript into a point in a vector space: transcripts dealing with similar issues have similar vectors; hence, the clustering step can identify clusters that share the same topics.</p></sec><sec id="s2-4-4"><title>Dimension Reduction</title><p>Dimension reduction transforms high-dimensional embeddings into a lower-dimensional representation that approximately preserves their geometric structure, thereby simplifying clustering and visualization. We projected the embeddings to a 5D space using UMAP (Uniform Manifold Approximation and Projection) [<xref ref-type="bibr" rid="ref53">53</xref>]. Reducing embeddings with UMAP before density-based clustering is standard practice in modern topic modeling and is the default pipeline in BERTopic [<xref ref-type="bibr" rid="ref21">21</xref>]. This step facilitates clustering, as we reduced the search space from 4096 to 5 dimensions. The number of components is not a sensitive design choice: the clinical associations are essentially unchanged when it is varied from 2 to 10. The choice of reduction method does matter. Replacing UMAP with a linear (principal component analysis) or alternative nonlinear (t-distributed stochastic neighbor embedding) reduction recovers weaker associations, because neither preserves the local neighborhood density that density-based clustering relies on (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). A neighborhood-preserving nonlinear reduction such as UMAP is therefore an important part of the method, whereas its exact dimensionality is not.</p></sec><sec id="s2-4-5"><title>Clustering</title><p>Clustering consisted of finding groups of patients with semantically similar transcripts. These transcripts may contain common topics that are clinically relevant. We extracted clusters using HDBSCAN (Hierarchical Density-Based Spatial Clustering of Applications With Noise) [<xref ref-type="bibr" rid="ref54">54</xref>], an algorithm that (1) requires no a priori specification of the number of clusters, (2) adapts to variable-density clusters, and (3) labels ambiguous points as noise.</p></sec><sec id="s2-4-6"><title>Cluster Description</title><p>We automatically labeled each cluster by randomly sampling 30 transcripts and asking a language model to summarize them, focusing on what is common between the transcripts. This sample size was chosen because (1) descriptions generated from 30 transcripts are reproducible across resampling, repeated generation, and prompt wording (<xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>), and (2) it approaches the maximum input that fits within the model&#x2019;s context window (40,960 tokens for Qwen3-14B). As clustered transcripts are semantically similar by construction, 30 samples typically capture the cluster&#x2019;s core themes without redundancy. The model received only the transcripts, so the generated descriptions are produced independently of any clinical score or demographic variable. As with the embedding model, we required an open-weight model under a permissive license, so that sensitive clinical speech was never sent to a third-party or commercial service. We used Qwen3-14B [<xref ref-type="bibr" rid="ref55">55</xref>], an instruction-tuned 14-billion-parameter model from a leading open-weight family, with strong multilingual coverage and a context window large enough to take the 30-transcript input in a single pass. We automatically obtained textual descriptions of clusters that contain several topics. This generative approach yields fine-grained topic descriptions automatically and at scale, going beyond keyword-based descriptions, which stay coarse, and manual coding, which does not scale. The prompt is provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p></sec><sec id="s2-4-7"><title>Qualitative Description</title><p>A board-certified psychiatrist (one of the coauthors) read the generated cluster descriptions as a coherence check, a qualitative reading rather than a clinical validation.</p></sec><sec id="s2-4-8"><title>Statistical Analysis</title><p>Each transcript is associated with a clinical score or diagnosis, such as a PHQ-9 score or a clinician-assigned depression diagnosis. Each cluster is therefore associated with a distribution of clinical scores. We performed statistical tests to determine whether there are statistical differences between the clusters based on their clinical scores. Transcripts that HDBSCAN labeled as noise were excluded from these cluster-level tests.</p><p>For continuous scores such as PHQ-9, we performed a Kruskal-Wallis H test to determine whether the clusters follow the same distribution or not. If we rejected the null hypothesis, we performed a pairwise Dunn test (with Benjamini-Hochberg corrections) to find significant differences between clusters. We reported effect sizes using eta-squared (&#x03B7;&#x00B2;=(H&#x2212;k+1)/(n&#x2212;k), where H is the Kruskal-Wallis statistic, k the number of clusters, and n the number of transcripts), interpreted following Cohen conventions: 0.01=small, 0.06=medium, and 0.14=large [<xref ref-type="bibr" rid="ref56">56</xref>]. We also report adjusted <italic>P</italic> values.</p><p>For categorical outcomes such as a clinician-assigned depression diagnosis, we tested whether the distribution of the outcome differed across clusters using a chi-square test of independence. If the omnibus test was significant, we performed pairwise chi-square tests with Benjamini-Hochberg corrections. We report effect sizes using Cram&#x00E9;r V and adjusted <italic>P</italic> values. For the small clinical cohorts, where sparse expected cell counts can violate the chi-square large-sample assumption, we also report the expected counts and confirm each principal association with an assumption-free Monte-Carlo permutation chi-square (<xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>).</p><p>We applied 2 Benjamini-Hochberg false-discovery-rate corrections. Within each significant omnibus test, the pairwise comparisons (pairwise Dunn or pairwise chi-square) were corrected together as one family. Across the question-by-clinical-score grid in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>, the omnibus <italic>P</italic> values were corrected jointly across all cells within a cohort. These omnibus tests and effect sizes are exploratory rather than prespecified: this study was designed to identify which questions and clinical constructs yield clusters associated with clinical status, not to test a small set of hypotheses fixed in advance.</p><p>We report effect sizes from the primary clustering, the single pipeline run on which all figures and cluster descriptions are based. Where reclustering is unstable, as for the Spanish suicide-risk association, we instead report the seed-averaged value and label the finding exploratory. <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref> shows these associations are robust to the clustering seed and reports a bootstrap 95% CI for every effect size stated in this paper. These CIs are conditional on the reported clustering: they quantify the sampling uncertainty of an effect size for a fixed partition and do not capture the additional uncertainty from rerunning the clustering or from choosing which interview questions to report. Uncertainty from reclustering is characterized separately by the seed-stability analysis (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>), and the choice of questions and scores is exploratory rather than prespecified. The CIs are therefore widest, and read most cautiously, in the small clinical cohorts. Throughout the Results section, a parenthetical range accompanying an effect size is its 95% CI (for example, &#x03B7;&#x00B2;=0.19, 0.17 to 0.24).</p></sec></sec><sec id="s2-5"><title>Ethical Considerations</title><sec id="s2-5-1"><title>Overview</title><p>All 4 cohorts used in this work had previously obtained informed consent from participants and ethical approval for speech data collection and secondary analysis.</p></sec><sec id="s2-5-2"><title>Data Governance</title><p>All computational processing used open-weight models (Whisper large-v3-turbo, Qwen3-Embedding-8B, and Qwen3-14B) that the authors ran themselves on controlled computing infrastructure. No transcripts, audio recordings, or derived data were transmitted to any third-party or external commercial API service at any stage of the pipeline. Data were stored without directly identifying information.</p></sec><sec id="s2-5-3"><title>French General Population Cohort</title><p>All participants provided informed consent in accordance with the Declaration of Helsinki, Good Clinical Practice guidelines, and local regulations. This study received approval from the French National Institutional Review Board (identifier 23.00748.OOO2L7#I). Data was securely stored without identifying information, and participants received a gift card for their time.</p></sec><sec id="s2-5-4"><title>Italian Clinical Population (Androids) Cohort</title><p>All participants volunteered and provided written informed consent in accordance with Italian and European Union data-protection law. The protocol was approved by the Ethics Committee of the Department of Psychology, Universit&#x00E0; degli Studi della Campania &#x201C;Luigi Vanvitelli&#x201D; (protocol 09/2016).</p></sec><sec id="s2-5-5"><title>Chinese Clinical Population (MODMA) Cohort</title><p>All recordings were collected under ethical guidelines approved by the Ethics Committee of the Second Affiliated Hospital of Lanzhou University. Participants provided written informed consent and received approximately US $16 as compensation.</p></sec><sec id="s2-5-6"><title>Spanish Clinical Population (VOCES) Cohort</title><p>All procedures were approved by the Ethics Committee of the University Hospital Fundaci&#x00F3;n Jim&#x00E9;nez D&#x00ED;az (PIC128-21_FJD) and complied with the Declaration of Helsinki. Participants provided written informed consent, and this study involved no costs or monetary compensation.</p></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview</title><p>We present results at 3 complementary levels: (1) quantitative identification of clusters with statistically significant differences in clinical scores and demographics, (2) automatically generated natural-language descriptions of cluster content (presented alongside quantitative results), and (3) a qualitative reading of the descriptions by a board-certified psychiatrist among the coauthors (qualitative description of clusters).</p></sec><sec id="s3-2"><title>Some Clusters Are Associated With High and Low Clinical Scores</title><p>Each transcript is associated with clinical scores. Each cluster of transcripts is thus a distribution of scores that can be compared to other distributions. Across all 4 cohorts, unsupervised clustering of transcript embeddings yielded between 4 and 26 distinct semantic clusters per cohort-question combination.</p><p>Cluster membership was significantly associated with clinical outcomes in the French, Italian, and Chinese cohorts, separating clusters with high and low clinical scores (<xref ref-type="fig" rid="figure1">Figures 1</xref>, <xref ref-type="fig" rid="figure4">4</xref>, and <xref ref-type="fig" rid="figure5">5</xref>). In the smaller Spanish cohort, the association was unstable across clustering seeds and is reported as exploratory, by its seed-averaged effect size (mean Cram&#x00E9;r V=0.27; <xref ref-type="fig" rid="figure2">Figure 2</xref>).</p><p>For the French general population cohort based on answers to &#x201C;Describe how you are feeling at the moment and how your sleep has been lately,&#x201D; among 26 identified clusters (n=1645 nonnoise transcripts), PHQ-9 scores differed significantly across clusters (&#x03B7;&#x00B2;=0.19, 0.17 to 0.24, <italic>P</italic>&#x003C;.001). Cluster 26 (n=37, age 25&#x00B1;9 years) exhibited the highest depression scores (PHQ-9 13.4&#x00B1;5.4) with LLM-generated descriptions highlighting sleep disturbances characterized by insomnia, frequent awakenings, and restless sleep, alongside pervasive anxiety, emotional instability, and self-esteem issues. In contrast, cluster 1 (n=92, age 39&#x00B1;19 years) showed the lowest depression scores (PHQ-9 2.6&#x00B1;2.2), with descriptions emphasizing consistent satisfaction with current well-being, good sleep quality, and general relaxation.</p><p>Intermediate-scoring clusters showed age-specific thematic content: cluster 10 (PHQ-9 5.8&#x00B1;4.3, age 69&#x00B1;15 years, n=34) contained descriptions of frequent nighttime urinary interruptions attributed to age-related conditions, while cluster 12 (PHQ-9 6.2&#x00B1;4.1, age 24&#x00B1;9 years, n=67) described stress related to academic examinations and significant life decisions.</p><p>We also report clusters with the highest and lowest clinical scores along with their generated descriptions for the clinical populations: Androids (<xref ref-type="fig" rid="figure5">Figure 5</xref>), MODMA (<xref ref-type="fig" rid="figure1">Figure 1</xref>), and VOCES (<xref ref-type="fig" rid="figure2">Figure 2</xref>).</p></sec><sec id="s3-3"><title>Some Questions Are Better Suited to Find Clinical Differences</title><p>We calculated effect sizes across questions and clinical scores (eg, sociodemographics such as age and education) for each cohort, with a short observation under each heatmap (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). We found that some questions are better suited to finding significant differences between clusters.</p><p>For example, for the French general population, the question &#x201C;Describe how you are feeling at the moment and how your sleep has been lately&#x201D; is the one with the highest effect sizes (eta-squared, &#x03B7;&#x00B2;) across clinical scores such as AIS (0.15, 0.13 to 0.20), BDI (0.16, 0.14 to 0.21), GAD-7 (0.16, 0.14 to 0.21), MFI (0.14, 0.12 to 0.19), and PHQ-9 (0.19, 0.17 to 0.24; <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). Every one of these associations was significant at <italic>P</italic>&#x003C;.001 and remained so after the Benjamini-Hochberg false-discovery-rate control across the effect-size grid (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>), consistent across clinical scores. The method captures how participants describe their sleep and mood, not just whether they report problems. For instance, clusters distinguished age-related nocturia from anxiety-driven insomnia, distinctions impossible with binary screening questions. We showed the distribution of the PHQ-9 score for this question (<xref ref-type="fig" rid="figure4">Figure 4</xref>).</p><p>In contrast, questions about past or future events yielded small eta-squared effect sizes. For example, topics extracted from &#x201C;describe a negative event that happened to you in the past&#x201D; showed: AIS (&#x03B7;&#x00B2;=0.01, 0.01 to 0.04), BDI (&#x03B7;&#x00B2;=0.01, 0.01 to 0.05), GAD-7 (&#x03B7;&#x00B2;=0.02, 0.02 to 0.06), MFI (&#x03B7;&#x00B2;=0.01, 0.01 to 0.05), and PHQ-9 (&#x03B7;&#x00B2;=0.03, 0.02 to 0.06). Similarly, questions about positive or negative future events did not produce clinically significant clusters. This question-selection pattern is robust to the clustering seed: across 10 random seeds, only the feelings-and-sleep question retains a substantial PHQ-9 association (mean &#x03B7;&#x00B2;=0.20), while the past- and future-event questions remain at or below &#x03B7;&#x00B2;=0.05 (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>).</p></sec><sec id="s3-4"><title>Age, Education, and Sex Are Systematically Associated With Topic Content Independently of Clinical Status</title><p>Beyond clinical scores, sociodemographic factors were also significantly associated with cluster membership.</p><p>This analysis was possible only in the French general population, the only cohort large enough to stratify participants across all subgroups and diagnoses. We examined whether sociodemographic factors (age, education, and sex) confound the association between cluster membership and clinical scores. Cluster 26 (highest PHQ-9; <xref ref-type="fig" rid="figure4">Figure 4</xref>) consisted predominantly of young adults (age 25&#x00B1;9 years), so we tested directly whether this cluster reflects depression-specific language or age-typical stressors within a population with depression.</p><p>To disentangle these effects directly, we fitted, for each clinical score, ordinary least-squares models of the score on cluster membership, on demographics (age, sex, and education), and on both, with SEs clustered by participant to account for repeated visits (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Cluster membership remained strongly associated with every score after adjustment and explained more variance than demographics: for PHQ-9 it accounted for 23.2% of the variance against 5.3% for demographics, and its partial &#x03B7;&#x00B2; changed little after adjustment (0.232 unadjusted to 0.220 adjusted, 0.188 to 0.279; joint test of cluster membership <italic>F</italic>=12.9, <italic>P</italic>&#x003C;.001). The same pattern held for all other scores and in a one-visit-per-participant sensitivity analysis. The demographic contribution is therefore genuine but partial, not one that accounts for the association between cluster membership and clinical scores entirely.</p><p>We also found that all questions led to significant differences between clusters for sociodemographics such as age, education, and sex (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). There are some exceptions: &#x201C;describe a negative event that happened to you in the past&#x201D; did not have a significant difference for education, and &#x201C;describe a positive event that happened to you in the past&#x201D; did not have a significant difference for sex.</p><p>The question &#x201C;describe your last 24 hours&#x201D; had the highest effect size for age (&#x03B7;&#x00B2;=0.27, 0.24 to 0.32) and sex (Cram&#x00E9;r V=0.22, 0.21 to 0.30). The following description corresponded to the cluster with the lowest mean age, mostly comprising young participants (age 22&#x00B1;5 years, n=37):</p><p>The individuals express involvement in professional or academic internships, collaboration with colleagues or mentors, and routine-based daily activities such as work tasks, social interactions, and personal responsibilities.</p><p>The following description corresponded to the cluster with the highest mean age, mostly comprising middle-aged and senior participants (age 55&#x00B1;15 years, n=56):</p><p>The individuals express common topics such as engaging in routine household tasks such as cooking, cleaning, and laundry, participating in family-centered activities such as shared meals and playing games with children, and allocating time to leisure pursuits, including watching television, movies, or using electronic devices for entertainment.</p><p>In both examples above, the cluster descriptions contain age-related topics.</p></sec><sec id="s3-5"><title>Qualitative Description of Clusters</title><p>The cluster descriptions are an interpretability layer, generated from the transcripts alone using a clinically agnostic prompt that never sees clinical scores or diagnoses. A board-certified psychiatrist (one of the coauthors) reviewed them and found them coherent with the clinical scores: across cohorts, clusters with higher scores carried descriptions of negative emotional states, functional impairment, and disrupted relationships, while lower-risk clusters emphasized positive coping strategies and supportive relationships. On the same clusters, the language-model descriptions state the clinical character of a cluster that keyword labels leave implicitly (<xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>). We separately examined whether the descriptions are grounded in the transcripts they summarize, rather than asserting unsupported content. Of the 40 cluster descriptions, 39 were faithful to their transcripts, the single exception imposing an organizing structure its cluster does not have (<xref ref-type="supplementary-material" rid="app9">Multimedia Appendix 9</xref>).</p><p>For example, in the Italian population (<xref ref-type="fig" rid="figure5">Figure 5</xref>), in the cluster with the highest proportion of depressive participants (cluster 5), the individuals expressed persistent struggles with psychiatric symptoms, exacerbated by significant life events, a sense of stagnation and purposelessness, and difficulties in professional and family functioning, whereas in the cluster with the lowest number of depressive participants (cluster 1), they highlighted the importance of family relationships, cultural affiliations, and personal interests.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><p>We presented a data-driven approach that discovers human-readable topic descriptions across general and clinical populations using language models. To our knowledge, this is the first study that automatically produces interpretable descriptions of clusters of spontaneous speech in different populations. We make 2 distinct claims. First, resting only on cluster membership (fixed by the embeddings and clustering of the transcripts), score-agnostic clusters of spontaneous speech are associated with clinical scores and separate validated scales in the French, Italian, and Chinese cohorts, with an exploratory association in the smaller Spanish cohort. Second, the language-model descriptions add interpretability, rendering those clusters in human-readable language more directly than keyword lists. Every statistical result belongs to the first claim and stands on its own, independent of how the clusters are described.</p><p>Three findings stand out. First, unsupervised clustering of transcript embeddings yields clusters that significantly associate with clinical scores in the French, Italian, and Chinese cohorts. Second, certain interview questions, for example &#x201C;describe how you are feeling at the moment and how your sleep has been lately,&#x201D; are more effective at yielding clusters associated with clinical status. Third, sociodemographic factors (age, education, and sex) systematically associate with cluster membership independently of clinical status, revealing that topic modeling captures both mental health phenomena and normative life circumstances, a distinction rarely examined in computational psychiatry. Beyond these, the pipeline automates the labor-intensive early stages of thematic analysis and surfaces, in a data-driven multilingual way, risk and protective topics consistent with those that resource-intensive expert consensus previously established by hand.</p><p>This third finding is central to interpreting the results. As age, sex, and education are themselves associated with cluster membership, a cluster associated with clinical status may be so in part through demographically patterned content. The clearest example is cluster 26 in the French cohort (highest PHQ-9), which is composed predominantly of young adults (age 25&#x00B1;9 years). Its &#x201C;depression-associated&#x201D; content may partly reflect age-typical stressors occurring in a depressed subgroup rather than depression-specific language per se. We therefore treat the demographic-confound analysis as a core result: a covariate-adjusted analysis (Results section and <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>) shows the association between cluster membership and clinical scores survives adjustment for age, sex, and education, and we return to its limits in the Limitations section.</p></sec><sec id="s4-2"><title>Content Not Captured by the Clinical Scales</title><p>Not every question or cluster was strongly associated with the clinical scales, and these weaker results are themselves informative. The questions about past or future events yielded stable clusters (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>) whose descriptions summarize life circumstances, yet their associations with the symptom scores were small (&#x03B7;&#x00B2; at or below 0.05; <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). Sociodemographic factors likewise shaped topic content independently of clinical status. In both cases, the pipeline recovered coherent content from speech that the symptom scales do not measure. We read this not as a shortcoming of the pipeline but as a reminder that a clinical scale quantifies a predefined set of symptoms and does not capture the full complexity of the lived experience of depression. Standard depression scales share little of their symptom content with one another [<xref ref-type="bibr" rid="ref57">57</xref>]. What patients consider recovery extends beyond symptom relief [<xref ref-type="bibr" rid="ref8">8</xref>]. First-person accounts of the lived experience of depression describe dimensions that no symptom scale measures [<xref ref-type="bibr" rid="ref14">14</xref>]. As the descriptions render this content in human-readable language rather than reducing it to a score, the approach reports content that a purely score-based analysis would leave out.</p></sec><sec id="s4-3"><title>Limitations</title><sec id="s4-3-1"><title>Overview</title><p>We set out the boundaries of this study and how they shape its interpretation.</p></sec><sec id="s4-3-2"><title>Sample Sizes and Representativeness</title><p>The clinical cohorts are modest in size (n=52&#x2010;116). We therefore tested cluster stability directly: a stability analysis across random seeds and minimum cluster sizes, covering every interview question in all 4 cohorts, together with internal cluster-validity indices for each cohort&#x2019;s principal clustering, is reported in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>. The French general population cohort, while larger (n=1809), over-represents highly educated participants. Larger-scale investigations with more balanced recruitment are needed to establish population-level generalizability.</p></sec><sec id="s4-3-3"><title>Confound Analysis</title><p>We performed the covariate-adjusted confound analysis only on the French general population cohort, as the clinical cohorts lacked sufficient sample sizes for it. In the French cohort, the association between cluster membership and clinical scores survived adjustment for age, sex, and education (Results section and <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Whether the same holds in the clinical samples remains open, because their size did not support the adjustment there. Extending covariate-adjusted or age-matched designs to the clinical cohorts is the next step. Future work should also incorporate demographic covariates directly into clustering.</p></sec><sec id="s4-3-4"><title>Heterogeneity of Clinical Measures</title><p>Our cohorts used different clinical instruments: PHQ-9 (self-reported depression severity), a clinician-assigned depression diagnosis, and the C-SSRS (suicide risk). These measure related but distinct constructs, limiting direct cross-cohort comparisons. This heterogeneity reflects the reality of working with independently collected clinical datasets, and it bounds the cross-cohort comparisons to the exploratory level.</p></sec><sec id="s4-3-5"><title>Transcription and Model Biases</title><p>Although we used state-of-the-art speech-to-text models, transcript noise remains a potential source of bias. Mental disorders can alter speech production (eg, reduced energy and dysfluencies) [<xref ref-type="bibr" rid="ref58">58</xref>], which may increase automatic speech recognition error rates that propagate to embeddings and clusters. We could not measure word or character error rates on our recordings, because no human reference transcripts were collected for any of the 4 cohorts. Published evaluations of the Whisper family report strong recognition accuracy for French, Italian, and Spanish and a higher error rate for Mandarin Chinese [<xref ref-type="bibr" rid="ref50">50</xref>]. A further concern is differential error. If transcription quality varies with clinical status, transcription error could itself contribute to the associations between cluster membership and the clinical scales. We have no human reference transcripts to test this directly, and because recognition is weakest for Chinese, the Chinese cohort warrants the most caution here. Creating reference transcripts for a small subset would permit direct error-rate measurement and is a worthwhile direction for future work. Additionally, the language models used to embed transcripts and generate cluster descriptions may encode cultural or demographic biases [<xref ref-type="bibr" rid="ref59">59</xref>]. LLMs also differ systematically in the values they prioritize in the text they generate [<xref ref-type="bibr" rid="ref60">60</xref>]. As all cluster descriptions in this study were generated by a single model, Qwen3-14B, they may reflect that model&#x2019;s value priorities rather than entirely neutral summaries of the transcripts.</p></sec><sec id="s4-3-6"><title>Clinical Validation of Cluster Descriptions</title><p>The cluster descriptions are an interpretability layer, and this study makes no claim about their clinical accuracy, completeness, or relevance. Establishing that would require independent, blinded evaluation by multiple clinicians with formal interrater reliability (Cohen &#x03BA; or the intraclass correlation coefficient), a distinct line of work. A board-certified psychiatrist read the descriptions as a coherence check. As all statistical associations are computed from cluster membership rather than from the descriptions, the findings do not depend on such clinical evaluation. The clusters that the pipeline recovers separate validated clinical scales in the French, Italian, and Chinese cohorts, and they are formed from the transcripts alone, without the clinical scores. The grounding of the descriptions in their transcripts (faithfulness), a comparison of each summary against its source text rather than a clinical judgment, is reported in <xref ref-type="supplementary-material" rid="app9">Multimedia Appendix 9</xref>, where 39 of 40 descriptions assert no content that their transcripts do not support and their stability across transcript sampling, repeated generation, and prompt wording in <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>, where they prove highly reproducible.</p></sec></sec><sec id="s4-4"><title>Comparison With Prior Work</title><sec id="s4-4-1"><title>Exploratory Cross-Cohort Observations</title><p>Across all 4 cohorts, clusters with higher depression or suicide risk scores consistently contained descriptions of negative emotional states, functional impairment, and disrupted relationships, while lower-risk clusters emphasized positive coping strategies, supportive relationships, and engagement in meaningful activities. We report this recurrence as a preliminary cross-cohort observation, not as evidence of cross-cultural generalizability, because the 4 datasets differ in language, population, clinical instruments, and interview questions; whether these markers generalize across linguistic and cultural contexts is a hypothesis for future studies with parallel data-collection protocols.</p><p>The descriptions also surfaced content specific to individual cohorts. In the Italian cohort, participants expressed strong regional identity and affiliation with Naples. In the Chinese cohort, family communication challenges and feelings of inadequacy featured prominently in higher-risk clusters. These cases show the pipeline captures cohort-specific themes a generic analysis might miss. Whether such content reflects broader cultural patterns or the cohorts&#x2019; particular recruitment and interview design cannot be determined from these nonparallel datasets, and would require dedicated cross-cultural studies with parallel protocols [<xref ref-type="bibr" rid="ref18">18</xref>].</p></sec><sec id="s4-4-2"><title>Risk and Protective Topics for Depression</title><p>We compared our generated topic descriptions with known risk and protective topics for depression studied in psychiatry research. A psychiatrist and a researcher in natural language processing reviewed the generated descriptions to relate them to risk and protective topics established in the literature.</p><p>Several clusters recapitulate established risk topics. Clusters with sleep disturbance or fatigue align with meta-analyses showing a bidirectional link between insomnia and depressive morbidity [<xref ref-type="bibr" rid="ref61">61</xref>]. Stressor-laden clusters (stress due to unemployment or financial difficulties) track established longitudinal associations between adverse psychosocial work environments or job loss and increased depression risk [<xref ref-type="bibr" rid="ref62">62</xref>]. Student-specific clusters may contain topics such as examination pressure that show associations between academic stressors and depressive symptoms [<xref ref-type="bibr" rid="ref63">63</xref>].</p><p>Other clusters map onto protective topics. Clusters emphasizing arts participation and creative activity correspond to the WHO scoping review concluding that arts engagement contributes to prevention and symptom reduction in mental health [<xref ref-type="bibr" rid="ref64">64</xref>]. Gardening and nature exposure clusters show consistency with evidence of small-to-moderate improvements in depressive symptoms from horticultural and green-space activities [<xref ref-type="bibr" rid="ref65">65</xref>]. Mentions of holidays or travel appear predominantly in low-severity clusters and are congruent with longitudinal work showing short-lived but reliable gains in mood and well-being during vacations [<xref ref-type="bibr" rid="ref66">66</xref>]. Finally, physical activity topics align with prospective meta-analyses, indicating that even subguideline levels of activity are associated with lower incident depression [<xref ref-type="bibr" rid="ref67">67</xref>].</p><p>Taken together, the fine-grained topics generated by the language model are consistent with core psychiatric constructs established in the literature, a qualitative correspondence rather than a statistical one. We found risk topics (eg, related to sleep disruption, workload, and financial stressors) and protective topics (eg, related to arts, nature, holidays as short-term relief, and physical activity) in a multilingual, data-driven fashion that yields human-readable descriptions while scaling beyond manual thematic analysis.</p></sec><sec id="s4-4-3"><title>Comparison With Keyword-Based Topic Modeling</title><p>Prior work on topic modeling in depression has relied on keyword-based approaches that surface topics as word lists. For instance, the study by Zhang et al [<xref ref-type="bibr" rid="ref28">28</xref>] applied BERTopic to free-response speech in the RADAR-MDD dataset and identified 29 topics represented by keywords such as "daughter," "son," "parents," "hospital," "operation," "pain," "sleep," "tired," and "rest." While such keywords indicate broad thematic areas, they lack the semantic context and narrative coherence that clinicians use when interpreting patient discourse. That study is the closest precedent to our work. It analyzes a single English-language dataset, represents each topic as a keyword list, and relates the topics to Patient Health Questionnaire-8 depression scores. Our study builds on this line of work and broadens it to 4 cohorts in 4 languages, including a large general-population sample and 3 clinical cohorts, and to several clinical constructs (depression, anxiety, insomnia, fatigue, and suicide risk).</p><p>Our approach produces qualitatively different outputs and requires no manual post hoc categorization: the language model directly produces interpretable summaries, whereas keyword-based approaches typically require researchers to manually group and label topics. To make this concrete on our own data, we labeled the same clusters in 2 ways, with keyword labels from a standard keyword-based topic model (BERTopic, which ranks the words most distinctive to each cluster after removing common stopwords in each language) and with our language-model descriptions. We hold the clustering fixed, so every statistical association is unchanged and only what a reader can take from the label differs. The descriptions state the clinical character in a single readable sentence and disambiguate context a bare keyword cannot, whereas keyword lists mix the theme with idiosyncratic or generic words that carry no clinical content. For example, the most depressed cluster in the Chinese cohort carries the keyword label "feeling," especially, "think," "before," "now," "nothing," "seem," "do not want," "life," which conveys little clinical content, whereas its language-model description names physical health issues and emotional struggles such as "depression," "anxiety," and "loss of motivation." In the Spanish sadness-and-coping cohort, the description reports self-inflicted physical pain as a maladaptive coping strategy that the cluster&#x2019;s keywords ("sad," "bad," "to feel," "sensation," "going out," "music," "week," "news," and "Madrid") do not surface. This contrast is not specific to our pipeline. Evaluations of topic models report that keyword-coherence metrics capture surface word statistics rather than the semantic quality a human reader needs, whereas language-model readings of topics align with human judgment [<xref ref-type="bibr" rid="ref68">68</xref>,<xref ref-type="bibr" rid="ref69">69</xref>]. The full side-by-side examples for every cohort, with the descriptions reproduced verbatim, are provided in <xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>.</p></sec></sec><sec id="s4-5"><title>Bridging Qualitative Research and Clinical Practice</title><p>Our introduction highlighted that manual thematic analysis, while clinically valuable, is too resource-intensive for routine care [<xref ref-type="bibr" rid="ref14">14</xref>]. The cost is concrete. Full thematic analysis of a single qualitative interview study can require more than 100 hours of researcher time [<xref ref-type="bibr" rid="ref70">70</xref>], and reliable manual coding further depends on trained coders and repeated double-coding to reach acceptable agreement [<xref ref-type="bibr" rid="ref71">71</xref>]. The present study shows that language models can generate cluster descriptions consistent with core psychiatric constructs identified in labor-intensive qualitative research. For instance, our clusters containing sleep disturbance, emotional dysregulation, and impaired functioning align with the phenomenological dimensions of depression (altered temporal experience, embodied distress, and existential emptiness) identified through expert consensus [<xref ref-type="bibr" rid="ref14">14</xref>].</p><p>This convergence indicates that automated methods can help bridge the gap between what qualitative analysis reveals as clinically important and what clinicians can feasibly assess at scale. Where the expert-consensus approach to identifying clinically validated themes required collaborative workshops and international expert panels [<xref ref-type="bibr" rid="ref14">14</xref>], our pipeline recovered comparable thematic structure automatically, across 4 languages and thousands of transcripts, without trained coders. It performs the labor-intensive early stages of thematic analysis&#x2014;familiarization with the material, initial coding, and theme generation&#x2014;in a single automated pass, while validation and clinical interpretation remain with the clinician. On our own data, the gain in speed and cost is concrete. On a single NVIDIA A100 GPU, the pipeline clustered and labeled a French interview question in minutes of compute, at a cloud-equivalent cost under US $1. The familiarization, initial coding, and theme-generation stages it automates would take an estimated 90 to 170 analyst-hours by hand, and roughly double that under the independent double-coding that reliable analysis requires [<xref ref-type="bibr" rid="ref71">71</xref>]. Across all 4 cohorts, the estimated manual effort runs to several hundred analyst-hours (<xref ref-type="supplementary-material" rid="app10">Multimedia Appendix 10</xref>). This is an illustrative comparison of scale rather than a claim of equivalence. The pipeline is an automated topic-modeling procedure of embedding, clustering, and language-model description, and it accelerates these early stages, not the interpretive coding and clinical judgment that a full manual thematic analysis brings. This pattern matches the broader literature: generative-AI thematic analysis reproduced most human-derived themes while reducing analyst time by about 97% [<xref ref-type="bibr" rid="ref72">72</xref>], with substantial reliability reported for automated content coding [<xref ref-type="bibr" rid="ref73">73</xref>]. Blinded comparisons find LLMs comparable to human analysts when applying predefined codes, with low hallucination rates [<xref ref-type="bibr" rid="ref74">74</xref>] and language-model theme summaries reaching substantial agreement with human-derived themes [<xref ref-type="bibr" rid="ref75">75</xref>]. Our approach identified clinically relevant content from speech records across diverse cohorts. The language model generated meaningful syntheses using clinically agnostic prompts, without pre-established theory or domain-specific information.</p><p>However, automated methods complement rather than replace clinical judgment. Expert clinical judgment remains essential for contextualizing algorithmically discovered patterns and for distinguishing clinically actionable insights from statistical artifacts.</p></sec><sec id="s4-6"><title>Potential Clinical Applications</title><p>This study establishes the methodological foundations. The clinical applications below are future possibilities. Our analyses show that cluster membership is associated with validated clinical scales in the French, Italian, and Chinese cohorts, but they do not establish screening validity, diagnostic accuracy, treatment-planning utility, or readiness for clinical decision support. Each of the directions below would require dedicated prospective validation. First, cluster-based analysis could in principle support screening and risk stratification: individuals whose speech patterns place them in high-severity clusters might be flagged for priority clinical review. Second, the fine-grained topic descriptions could inform personalized treatment planning: a patient whose cluster emphasizes academic stress and sleep disturbance may benefit from different interventions than one whose cluster highlights social isolation and hopelessness. Third, tracking cluster membership or distance-to-cluster-centroid over time could enable treatment monitoring, providing objective measures of thematic shifts during therapy.</p><p>Critically, such applications require interpretability. Clinicians need to understand why a patient was assigned to a particular cluster to trust and act on algorithmic recommendations. Our approach addresses this need by generating human-readable descriptions rather than opaque embeddings or keyword lists. This interpretability is essential for developing safe and trustworthy clinical decision support systems based on speech or language analysis. Future work should evaluate whether cluster descriptions improve clinician decision-making in controlled settings.</p></sec><sec id="s4-7"><title>Conclusions</title><p>Systematic analysis of spontaneous speech at scale has been impractical, and existing automated methods describe topics with keyword lists that miss clinical nuance. The pipeline we introduce automates the labor-intensive early stages of thematic analysis (familiarization, initial coding, and theme generation) on spontaneous speech in 4 languages, in minutes of single-GPU compute per interview question, at a cost under US $1. It uses a language model to render each cluster as a description a clinician can read directly. In the French, Italian, and Chinese cohorts, score-agnostic clusters were significantly associated with clinical scores, with moderate to large effects (for example, PHQ-9 &#x03B7;&#x00B2;=0.19, and Cram&#x00E9;r V=0.74 for clinician-diagnosed depression). The association in the smaller Spanish cohort was exploratory. The descriptions that label the clusters stayed grounded in the transcripts and were reproducible.</p><p>Two findings carry practical weight. Question choice matters: a single question about current feelings and sleep separated clusters by symptom severity, whereas questions about past or future events did not. Sociodemographic factors shaped topic content independently of clinical status, a confound rarely examined in computational psychiatry, although the association between cluster membership and clinical scores survived adjustment for age, sex, and education.</p><p>As the 4 datasets are independent and were not collected in parallel, the cross-cohort patterns are preliminary rather than evidence of cross-cultural generalizability. Screening, treatment-planning, and decision-support uses remain to be established through dedicated prospective study. By making patient narrative analyzable at scale in readable language, the approach can augment, not replace, the clinician&#x2019;s judgment.</p></sec></sec></body><back><ack><p>The authors thank the participants of the 4 cohorts for contributing their time and speech data, and the clinical teams at the collaborating institutions for their support in recruitment and data collection. The authors also acknowledge the EuroHPC Joint Undertaking for awarding this project access to the EuroHPC supercomputer LEONARDO, hosted by CINECA (Italy) and the LEONARDO consortium, through a EuroHPC Development Access call (project EHPC-DEV-2026D02-260). This work was also performed using high-performance computing resources from GENCI-IDRIS (AD010315777R1). During the preparation of this work, the authors used several computational methods as part of the research methodology described in the Methods section, and not in the writing of this paper. Whisper large-v3-turbo transcribed the speech recordings. Qwen3-Embedding-8B encoded each transcript as a numerical vector. UMAP (Uniform Manifold Approximation and Projection) reduced the dimensionality of those vectors, and HDBSCAN (Hierarchical Density-Based Spatial Clustering of Applications With Noise) grouped them into clusters. UMAP and HDBSCAN are classical algorithms, not language models. Qwen3-14B, a generative large language model, produced the natural-language cluster descriptions and is the only one of these methods that generates new text. No generative AI or AI-assisted technologies were used to write this paper. The cluster descriptions that Qwen3-14B produced are reported as a research result and are reproduced verbatim in <xref ref-type="supplementary-material" rid="app8">Multimedia Appendices 8</xref> and <xref ref-type="supplementary-material" rid="app9">9</xref>. The authors reviewed all model-generated content and take full responsibility for the content of this published paper.</p></ack><notes><sec><title>Funding</title><p>This project has received funding from the European Innovation Council under the European Union&#x2019;s Horizon Europe research and innovation programme (101295925). The funder had no involvement in the study design, data collection, analysis, interpretation, or the writing of this paper.</p></sec><sec><title>Data Availability</title><p>Due to the sensitive nature of clinical speech data containing personal health information, raw data cannot be shared publicly. All methods, model specifications, hyperparameters, and prompts are fully described in the Methods section and appendices to enable reproducibility. Researchers seeking collaboration or data access for replication purposes may contact the corresponding author.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: GC, XNC, RR</p><p>Data curation: GC, SdL-M</p><p>Formal analysis: GC, SBG</p><p>Investigation: GC, SBG, SdL-M, EB-G</p><p>Methodology: GC, XNC, EB-G, RR</p><p>Project administration: XNC, RR</p><p>Resources: XNC, SdL-M, EB-G, RR</p><p>Software: GC</p><p>Supervision: XNC, EB-G, RR</p><p>Validation: SBG</p><p>Visualization: GC</p><p>Writing &#x2013; original draft: GC</p><p>Writing &#x2013; review &#x0026; editing: GC, SBG, XNC, SdL-M, EB-G, RR</p></fn><fn fn-type="conflict"><p>Callyope developed the mobile data-collection platform used to record the French general-population cohort. RR and XNC (founders) and GC (research scientist) hold equity in Callyope. SBG served as a psychiatrist consultant for the company. The remaining authors, SdL-M and EB-G, declare no competing interests. The Italian, Chinese, and Spanish cohorts are independently collected, publicly documented datasets in which Callyope had no role.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AIS</term><def><p>Athens Insomnia Scale</p></def></def-item><def-item><term id="abb2">BDI</term><def><p>Beck Depression Inventory</p></def></def-item><def-item><term id="abb3">C-SSRS</term><def><p>Columbia Suicide Severity Rating Scale</p></def></def-item><def-item><term id="abb4"><italic>DSM</italic></term><def><p><italic>Diagnostic and Statistical Manual of Mental Disorders</italic></p></def></def-item><def-item><term id="abb5"><italic>DSM-5</italic></term><def><p><italic>Diagnostic and Statistical Manual of Mental Disorders</italic> (Fifth Edition)</p></def></def-item><def-item><term id="abb6"><italic>DSM-IV</italic></term><def><p><italic>Diagnostic and Statistical Manual of Mental Disorders</italic> (Fourth Edition)</p></def></def-item><def-item><term id="abb7">GAD-7</term><def><p>Generalized Anxiety Disorder 7-item scale</p></def></def-item><def-item><term id="abb8">HDBSCAN</term><def><p>Hierarchical Density-Based Spatial Clustering of Applications With Noise</p></def></def-item><def-item><term id="abb9">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb10">MDD</term><def><p>major depressive disorder</p></def></def-item><def-item><term id="abb11">MFI</term><def><p>Multidimensional Fatigue Inventory</p></def></def-item><def-item><term id="abb12">MODMA</term><def><p>multi-modal open dataset for mental-disorder analysis</p></def></def-item><def-item><term id="abb13">PHQ-9</term><def><p>Patient Health Questionnaire-9</p></def></def-item><def-item><term id="abb14">UMAP</term><def><p>Uniform Manifold Approximation and Projection</p></def></def-item><def-item><term id="abb15">VOCES</term><def><p>Spanish clinical cohort for suicide-risk research</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bromet</surname><given-names>E</given-names> </name><name name-style="western"><surname>Andrade</surname><given-names>LH</given-names> </name><name name-style="western"><surname>Hwang</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Cross-national epidemiology of DSM-IV major depressive episode</article-title><source>BMC Med</source><year>2011</year><month>07</month><day>26</day><volume>9</volume><issue>1</issue><fpage>90</fpage><pub-id pub-id-type="doi">10.1186/1741-7015-9-90</pub-id><pub-id pub-id-type="medline">21791035</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Herrman</surname><given-names>H</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>V</given-names> </name><name name-style="western"><surname>Kieling</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Time for united action on depression: a Lancet-World Psychiatric Association Commission</article-title><source>Lancet</source><year>2022</year><month>03</month><day>5</day><volume>399</volume><issue>10328</issue><fpage>957</fpage><lpage>1022</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(21)02141-3</pub-id><pub-id pub-id-type="medline">35180424</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Regier</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Kuhl</surname><given-names>EA</given-names> </name><name name-style="western"><surname>Kupfer</surname><given-names>DJ</given-names> </name></person-group><article-title>The DSM-5: classification and criteria changes</article-title><source>World Psychiatry</source><year>2013</year><month>06</month><volume>12</volume><issue>2</issue><fpage>92</fpage><lpage>98</lpage><pub-id pub-id-type="doi">10.1002/wps.20050</pub-id><pub-id pub-id-type="medline">23737408</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aboraya</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rankin</surname><given-names>E</given-names> </name><name name-style="western"><surname>France</surname><given-names>C</given-names> </name><name name-style="western"><surname>El-Missiry</surname><given-names>A</given-names> </name><name name-style="western"><surname>John</surname><given-names>C</given-names> </name></person-group><article-title>The reliability of psychiatric diagnosis revisited: the clinician&#x2019;s guide to improve the reliability of psychiatric diagnosis</article-title><source>Psychiatry (Edgmont)</source><year>2006</year><month>01</month><volume>3</volume><issue>1</issue><fpage>41</fpage><lpage>50</lpage><pub-id pub-id-type="medline">21103149</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aboraya</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nasrallah</surname><given-names>HA</given-names> </name><name name-style="western"><surname>Elswick</surname><given-names>DE</given-names> </name><etal/></person-group><article-title>Measurement-based care in psychiatry-past, present, and future</article-title><source>Innov Clin Neurosci</source><year>2018</year><month>11</month><day>1</day><volume>15</volume><issue>11-12</issue><fpage>13</fpage><lpage>26</lpage><pub-id pub-id-type="medline">30834167</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Armstrong</surname><given-names>N</given-names> </name><name name-style="western"><surname>Byrom</surname><given-names>N</given-names> </name></person-group><article-title>An anthropological critique of psychiatric rating scales</article-title><source>BJPsych advances</source><year>2025</year><month>03</month><volume>31</volume><issue>2</issue><fpage>73</fpage><lpage>81</lpage><pub-id pub-id-type="doi">10.1192/bja.2024.60</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stanghellini</surname><given-names>G</given-names> </name><name name-style="western"><surname>Ballerini</surname><given-names>M</given-names> </name></person-group><article-title>Qualitative analysis. Its use in psychopathological research</article-title><source>Acta Psychiatr Scand</source><year>2008</year><month>03</month><volume>117</volume><issue>3</issue><fpage>161</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1111/j.1600-0447.2007.01139.x</pub-id><pub-id pub-id-type="medline">18271797</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chevance</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ravaud</surname><given-names>P</given-names> </name><name name-style="western"><surname>Tomlinson</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Identifying outcomes for depression that matter to patients, informal caregivers, and health-care professionals: qualitative content analysis of a large international online survey</article-title><source>Lancet Psychiatry</source><year>2020</year><month>08</month><volume>7</volume><issue>8</issue><fpage>692</fpage><lpage>702</lpage><pub-id pub-id-type="doi">10.1016/S2215-0366(20)30191-7</pub-id><pub-id pub-id-type="medline">32711710</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="web"><article-title>Manufacturer and user facility device experience (MAUDE) database</article-title><source>US Food and Drug Administration</source><access-date>2026-07-31</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.accessdata.fda.gov/scripts/cdrh/cfdocs/cfMAUDE/search.cfm">https://www.accessdata.fda.gov/scripts/cdrh/cfdocs/cfMAUDE/search.cfm</ext-link></comment></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Muralidharan</surname><given-names>V</given-names> </name><name name-style="western"><surname>Adewale</surname><given-names>BA</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>CJ</given-names> </name><etal/></person-group><article-title>A scoping review of reporting gaps in FDA-approved AI medical devices</article-title><source>NPJ Digit Med</source><year>2024</year><month>10</month><day>3</day><volume>7</volume><issue>1</issue><fpage>273</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01270-x</pub-id><pub-id pub-id-type="medline">39362934</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cummins</surname><given-names>N</given-names> </name><name name-style="western"><surname>Scherer</surname><given-names>S</given-names> </name><name name-style="western"><surname>Krajewski</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schnieder</surname><given-names>S</given-names> </name><name name-style="western"><surname>Epps</surname><given-names>J</given-names> </name><name name-style="western"><surname>Quatieri</surname><given-names>TF</given-names> </name></person-group><article-title>A review of depression and suicide risk assessment using speech analysis</article-title><source>Speech Commun</source><year>2015</year><month>07</month><volume>71</volume><fpage>10</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1016/j.specom.2015.03.004</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ehlen</surname><given-names>F</given-names> </name><name name-style="western"><surname>Montag</surname><given-names>C</given-names> </name><name name-style="western"><surname>Leopold</surname><given-names>K</given-names> </name><name name-style="western"><surname>Heinz</surname><given-names>A</given-names> </name></person-group><article-title>Linguistic findings in persons with schizophrenia-a review of the current literature</article-title><source>Front Psychol</source><year>2023</year><month>11</month><volume>14</volume><fpage>1287706</fpage><pub-id pub-id-type="doi">10.3389/fpsyg.2023.1287706</pub-id><pub-id pub-id-type="medline">38078276</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Trifu</surname><given-names>RN</given-names> </name><name name-style="western"><surname>Neme&#x0219;</surname><given-names>B</given-names> </name><name name-style="western"><surname>Herta</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Bodea-Hategan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Tala&#x0219;</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Coman</surname><given-names>H</given-names> </name></person-group><article-title>Linguistic markers for major depressive disorder: a cross-sectional study using an automated procedure</article-title><source>Front Psychol</source><year>2024</year><month>03</month><volume>15</volume><fpage>1355734</fpage><pub-id pub-id-type="doi">10.3389/fpsyg.2024.1355734</pub-id><pub-id pub-id-type="medline">38510303</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fusar-Poli</surname><given-names>P</given-names> </name><name name-style="western"><surname>Estrad&#x00E9;</surname><given-names>A</given-names> </name><name name-style="western"><surname>Stanghellini</surname><given-names>G</given-names> </name><etal/></person-group><article-title>The lived experience of depression: a bottom-up review co-written by experts by experience and academics</article-title><source>World Psychiatry</source><year>2023</year><month>10</month><volume>22</volume><issue>3</issue><fpage>352</fpage><lpage>365</lpage><pub-id pub-id-type="doi">10.1002/wps.21111</pub-id><pub-id pub-id-type="medline">37713566</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hallford</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Rusanov</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yeow</surname><given-names>JJE</given-names> </name><name name-style="western"><surname>Barry</surname><given-names>TJ</given-names> </name></person-group><article-title>Overgeneral and specific autobiographical memory predict the course of depression: an updated meta-analysis</article-title><source>Psychol Med</source><year>2021</year><month>04</month><volume>51</volume><issue>6</issue><fpage>909</fpage><lpage>926</lpage><pub-id pub-id-type="doi">10.1017/S0033291721001343</pub-id><pub-id pub-id-type="medline">33875023</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Cortal</surname><given-names>G</given-names> </name><name name-style="western"><surname>Finkel</surname><given-names>A</given-names> </name><name name-style="western"><surname>Paroubek</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ye</surname><given-names>L</given-names> </name></person-group><article-title>Emotion recognition based on psychological components in guided narratives for emotion regulation</article-title><conf-name>Proceedings of the 7th Joint SIGHUM Workshop on Computational Linguistics for Cultural Heritage, Social Sciences, Humanities and Literature</conf-name><conf-date>May 5-7, 2023</conf-date><conf-loc>Dubrovnik, Croatia</conf-loc><fpage>72</fpage><lpage>81</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.latechclfl-1.8</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Cortal</surname><given-names>G</given-names> </name><name name-style="western"><surname>Finkel</surname><given-names>A</given-names> </name></person-group><article-title>Formalizing style in personal narratives</article-title><conf-name>Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 4-9, 2025</conf-date><conf-loc>Suzhou, China</conf-loc><fpage>7311</fpage><lpage>7326</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.emnlp-main.371</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lehti</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hammarstr&#x00F6;m</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mattsson</surname><given-names>B</given-names> </name></person-group><article-title>Recognition of depression in people of different cultures: a qualitative study</article-title><source>BMC Fam Pract</source><year>2009</year><month>07</month><day>27</day><volume>10</volume><issue>1</issue><fpage>53</fpage><pub-id pub-id-type="doi">10.1186/1471-2296-10-53</pub-id><pub-id pub-id-type="medline">19635159</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Braun</surname><given-names>V</given-names> </name><name name-style="western"><surname>Clarke</surname><given-names>V</given-names> </name></person-group><article-title>Using thematic analysis in psychology</article-title><source>Qual Res Psychol</source><year>2006</year><month>01</month><volume>3</volume><issue>2</issue><fpage>77</fpage><lpage>101</lpage><pub-id pub-id-type="doi">10.1191/1478088706qp063oa</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Blei</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>AY</given-names> </name><name name-style="western"><surname>Jordan</surname><given-names>MI</given-names> </name></person-group><article-title>Latent Dirichlet allocation</article-title><source>J Mach Learn Res</source><year>2003</year><access-date>2026-07-31</access-date><volume>3</volume><issue>Jan</issue><fpage>993</fpage><lpage>1022</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://jmlr.csail.mit.edu/papers/v3/blei03a.html">https://jmlr.csail.mit.edu/papers/v3/blei03a.html</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Grootendorst</surname><given-names>M</given-names> </name></person-group><article-title>BERTopic: neural topic modeling with a class-based TF-IDF procedure</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 11, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2203.05794</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pham</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Hoyle</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>S</given-names> </name><name name-style="western"><surname>Resnik</surname><given-names>P</given-names> </name><name name-style="western"><surname>Iyyer</surname><given-names>M</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>H</surname><given-names>G</given-names> </name><name name-style="western"><surname>S</surname><given-names>B</given-names> </name></person-group><article-title>TopicGPT: a prompt-based topic modeling framework</article-title><conf-name>Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics</conf-name><conf-date>Jun 16-21, 2024</conf-date><conf-loc>Mexico City, Mexico</conf-loc><fpage>2956</fpage><lpage>2984</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.naacl-long.164</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Mu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>C</given-names> </name><name name-style="western"><surname>Bontcheva</surname><given-names>K</given-names> </name><name name-style="western"><surname>Song</surname><given-names>X</given-names> </name></person-group><article-title>Large language models offer an alternative to the traditional approach of topic modelling</article-title><conf-name>Joint International Conference on Computational Linguistics, Language Resources and Evaluation</conf-name><conf-date>May 20-25, 2024</conf-date><conf-loc>Turin, Italy</conf-loc><fpage>10160</fpage><lpage>10171</lpage><pub-id pub-id-type="doi">10.63317/2x489fw7wi5m</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Carron-Arthur</surname><given-names>B</given-names> </name><name name-style="western"><surname>Reynolds</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bennett</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bennett</surname><given-names>A</given-names> </name><name name-style="western"><surname>Griffiths</surname><given-names>KM</given-names> </name></person-group><article-title>What&#x2019;s all the talk about? Topic modelling in a mental health internet support group</article-title><source>BMC Psychiatry</source><year>2016</year><month>10</month><day>28</day><volume>16</volume><issue>1</issue><fpage>367</fpage><pub-id pub-id-type="doi">10.1186/s12888-016-1073-5</pub-id><pub-id pub-id-type="medline">27793131</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leung</surname><given-names>YT</given-names> </name><name name-style="western"><surname>Khalvati</surname><given-names>F</given-names> </name></person-group><article-title>Exploring COVID-19-related stressors: topic modeling study</article-title><source>J Med Internet Res</source><year>2022</year><month>07</month><day>13</day><volume>24</volume><issue>7</issue><fpage>e37142</fpage><pub-id pub-id-type="doi">10.2196/37142</pub-id><pub-id pub-id-type="medline">35731966</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Boettcher</surname><given-names>N</given-names> </name></person-group><article-title>Studies of depression and anxiety using Reddit as a data source: scoping review</article-title><source>JMIR Ment Health</source><year>2021</year><month>11</month><day>25</day><volume>8</volume><issue>11</issue><fpage>e29487</fpage><pub-id pub-id-type="doi">10.2196/29487</pub-id><pub-id pub-id-type="medline">34842560</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Shan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Topic modeling and content analysis of people&#x2019;s anxiety-related concerns raised on a computer-mediated health platform</article-title><source>Sci Rep</source><year>2024</year><month>11</month><day>11</day><volume>14</volume><issue>1</issue><fpage>27520</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-79164-x</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Folarin</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Dineley</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Identifying depression-related topics in smartphone-collected free-response speech recordings using an automatic speech recognition system and a deep learning topic model</article-title><source>J Affect Disord</source><year>2024</year><month>06</month><volume>355</volume><fpage>40</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2024.03.106</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Timakum</surname><given-names>T</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name></person-group><article-title>Identifying mental health discussion topic in social media community: subreddit of bipolar disorder analysis</article-title><source>Front Res Metrics Anal</source><year>2023</year><volume>8</volume><fpage>1243407</fpage><pub-id pub-id-type="doi">10.3389/frma.2023.1243407</pub-id><pub-id pub-id-type="medline">38025958</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Johannsen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hovy</surname><given-names>D</given-names> </name><name name-style="western"><surname>S&#x00F8;gaard</surname><given-names>A</given-names> </name></person-group><article-title>Cross-lingual syntactic variation over age and gender</article-title><conf-name>Proceedings of the Nineteenth Conference on Computational Natural Language Learning</conf-name><conf-date>Jul 30-31, 2015</conf-date><conf-loc>Beijing, China</conf-loc><fpage>103</fpage><lpage>112</lpage><pub-id pub-id-type="doi">10.18653/v1/K15-1011</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Preo&#x0163;iuc-Pietro</surname><given-names>D</given-names> </name><name name-style="western"><surname>Eichstaedt</surname><given-names>J</given-names> </name><name name-style="western"><surname>Park</surname><given-names>G</given-names> </name><etal/></person-group><article-title>The role of personality, age, and gender in tweeting about mental illness</article-title><conf-name>Proceedings of the 2nd Workshop on Computational Linguistics and Clinical Psychology</conf-name><conf-date>Jun 5, 2015</conf-date><conf-loc>Denver, Colorado</conf-loc><fpage>21</fpage><lpage>30</lpage><pub-id pub-id-type="doi">10.3115/v1/W15-1203</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rai</surname><given-names>S</given-names> </name><name name-style="western"><surname>Stade</surname><given-names>EC</given-names> </name><name name-style="western"><surname>Giorgi</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Key language markers of depression on social media depend on race</article-title><source>Proc Natl Acad Sci U S A</source><year>2024</year><month>04</month><day>2</day><volume>121</volume><issue>14</issue><fpage>e2319837121</fpage><pub-id pub-id-type="doi">10.1073/pnas.2319837121</pub-id><pub-id pub-id-type="medline">38530887</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Rosario</surname><given-names>C</given-names> </name><collab>The Nueva School, San Mateo, USA</collab></person-group><article-title>Age-specific linguistic features of depression via social media</article-title><conf-name>The 8th Student Research Workshop associated with the International Conference RANLP 2023</conf-name><conf-date>Sep 4-6, 2023</conf-date><pub-id pub-id-type="doi">10.26615/issn.2603-2821.2023_004</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Mo&#x00DF;burger</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wende</surname><given-names>F</given-names> </name><name name-style="western"><surname>Brinkmann</surname><given-names>K</given-names> </name><name name-style="western"><surname>Schmidt</surname><given-names>T</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>AZ</surname><given-names>K</given-names> </name><name name-style="western"><surname>I</surname><given-names>F</given-names> </name></person-group><article-title>Exploring online depression forums via text mining: a comparison of reddit and a curated online forum</article-title><access-date>2026-07-29</access-date><conf-name>Proceedings of the Fifth Social Media Mining for Health Applications Workshop &#x0026; Shared Task</conf-name><conf-date>Dec 12, 2020</conf-date><fpage>70</fpage><lpage>81</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2020.smm4h-1.11/">https://aclanthology.org/2020.smm4h-1.11/</ext-link></comment></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Low</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Rumker</surname><given-names>L</given-names> </name><name name-style="western"><surname>Talkar</surname><given-names>T</given-names> </name><name name-style="western"><surname>Torous</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cecchi</surname><given-names>G</given-names> </name><name name-style="western"><surname>Ghosh</surname><given-names>SS</given-names> </name></person-group><article-title>Natural language processing reveals vulnerable mental health support groups and heightened health anxiety on Reddit during COVID-19: observational study</article-title><source>J Med Internet Res</source><year>2020</year><month>10</month><day>12</day><volume>22</volume><issue>10</issue><fpage>e22635</fpage><pub-id pub-id-type="doi">10.2196/22635</pub-id><pub-id pub-id-type="medline">32936777</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Raihan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Puspo</surname><given-names>SSC</given-names> </name><name name-style="western"><surname>Farabi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bucur</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Ranasinghe</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zampieri</surname><given-names>M</given-names> </name></person-group><article-title>MentalHelp: a multi-task dataset for mental health in social media</article-title><conf-name>Joint International Conference on Computational Linguistics, Language Resources and Evaluation</conf-name><conf-date>May 20-25, 2024</conf-date><conf-loc>Turin, Italy</conf-loc><fpage>11196</fpage><lpage>11203</lpage><pub-id pub-id-type="doi">10.63317/372jegsog7nt</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Harrigian</surname><given-names>K</given-names> </name><name name-style="western"><surname>Aguirre</surname><given-names>C</given-names> </name><name name-style="western"><surname>Dredze</surname><given-names>M</given-names> </name></person-group><article-title>Do models of mental health based on social media data generalize?</article-title><conf-name>Findings of the Association for Computational Linguistics</conf-name><conf-date>Nov 16-20, 2020</conf-date><fpage>3774</fpage><lpage>3788</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.findings-emnlp.337</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levis</surname><given-names>B</given-names> </name><name name-style="western"><surname>Benedetti</surname><given-names>A</given-names> </name><name name-style="western"><surname>Thombs</surname><given-names>BD</given-names> </name><collab>DEPRESsion Screening Data (DEPRESSD) Collaboration</collab></person-group><article-title>Accuracy of Patient Health Questionnaire-9 (PHQ-9) for screening to detect major depression: individual participant data meta-analysis</article-title><source>BMJ</source><year>2019</year><month>04</month><day>9</day><volume>365</volume><issue>365</issue><fpage>l1476</fpage><pub-id pub-id-type="doi">10.1136/bmj.l1476</pub-id><pub-id pub-id-type="medline">30967483</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beck</surname><given-names>AT</given-names> </name><name name-style="western"><surname>Ward</surname><given-names>CH</given-names> </name><name name-style="western"><surname>Mendelson</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mock</surname><given-names>J</given-names> </name><name name-style="western"><surname>Erbaugh</surname><given-names>J</given-names> </name></person-group><article-title>An inventory for measuring depression</article-title><source>Arch Gen Psychiatry</source><year>1961</year><month>06</month><volume>4</volume><issue>6</issue><fpage>561</fpage><lpage>571</lpage><pub-id pub-id-type="doi">10.1001/archpsyc.1961.01710120031004</pub-id><pub-id pub-id-type="medline">13688369</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Okajima</surname><given-names>I</given-names> </name><name name-style="western"><surname>Miyamoto</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ubara</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Evaluation of severity levels of the Athens Insomnia Scale based on the criterion of Insomnia Severity Index</article-title><source>Int J Environ Res Public Health</source><year>2020</year><month>11</month><day>26</day><volume>17</volume><issue>23</issue><fpage>8789</fpage><pub-id pub-id-type="doi">10.3390/ijerph17238789</pub-id><pub-id pub-id-type="medline">33256097</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Smets</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Garssen</surname><given-names>B</given-names> </name><name name-style="western"><surname>Bonke</surname><given-names>B</given-names> </name><name name-style="western"><surname>De Haes</surname><given-names>JC</given-names> </name></person-group><article-title>The Multidimensional Fatigue Inventory (MFI) psychometric qualities of an instrument to assess fatigue</article-title><source>J Psychosom Res</source><year>1995</year><month>04</month><volume>39</volume><issue>3</issue><fpage>315</fpage><lpage>325</lpage><pub-id pub-id-type="doi">10.1016/0022-3999(94)00125-o</pub-id><pub-id pub-id-type="medline">7636775</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Tao</surname><given-names>F</given-names> </name><name name-style="western"><surname>Esposito</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vinciarelli</surname><given-names>A</given-names> </name></person-group><article-title>The Androids Corpus: a new publicly available benchmark for speech based depression detection</article-title><conf-name>INTERSPEECH 2023</conf-name><conf-date>Aug 20-24, 2026</conf-date><conf-loc>Dublin, Ireland</conf-loc><fpage>4149</fpage><lpage>4153</lpage><pub-id pub-id-type="doi">10.21437/Interspeech.2023-894</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cai</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yuan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>A multi-modal open dataset for mental-disorder analysis</article-title><source>Sci Data</source><year>2022</year><month>04</month><day>19</day><volume>9</volume><issue>1</issue><fpage>178</fpage><pub-id pub-id-type="doi">10.1038/s41597-022-01211-x</pub-id><pub-id pub-id-type="medline">35440583</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lecrubier</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sheehan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Weiller</surname><given-names>E</given-names> </name><etal/></person-group><article-title>The MINI International Neuropsychiatric Interview (MINI). a short diagnostic structured interview: reliability and validity according to the CIDI</article-title><source>Eur Psychiatr</source><year>1997</year><volume>12</volume><issue>5</issue><fpage>224</fpage><lpage>231</lpage><pub-id pub-id-type="doi">10.1016/S0924-9338(97)83296-8</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="book"><source>Diagnostic and Statistical Manual of Mental Disorders: DSM-5</source><year>2013</year><edition>5</edition><publisher-name>American Psychiatric Association</publisher-name><pub-id pub-id-type="doi">10.1176/appi.books.9780890425596</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hamilton</surname><given-names>M</given-names> </name></person-group><article-title>A rating scale for depression</article-title><source>J Neurol Neurosurg Psychiatry</source><year>1960</year><month>02</month><volume>23</volume><issue>1</issue><fpage>56</fpage><lpage>62</lpage><pub-id pub-id-type="doi">10.1136/jnnp.23.1.56</pub-id><pub-id pub-id-type="medline">14399272</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Leon-Martinez</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ruiz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Parra-Vargas</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Virtual reality and speech analysis for the assessment of impulsivity and decision-making: protocol for a comparison with neuropsychological tasks and self-administered questionnaires</article-title><source>BMJ Open</source><year>2022</year><month>07</month><day>13</day><volume>12</volume><issue>7</issue><fpage>e058486</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2021-058486</pub-id><pub-id pub-id-type="medline">35831051</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Posner</surname><given-names>K</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>GK</given-names> </name><name name-style="western"><surname>Stanley</surname><given-names>B</given-names> </name><etal/></person-group><article-title>The Columbia-Suicide Severity Rating Scale: initial validity and internal consistency findings from three multisite studies with adolescents and adults</article-title><source>Am J Psychiatry</source><year>2011</year><month>12</month><volume>168</volume><issue>12</issue><fpage>1266</fpage><lpage>1277</lpage><pub-id pub-id-type="doi">10.1176/appi.ajp.2011.10111704</pub-id><pub-id pub-id-type="medline">22193671</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Al-Halab&#x00ED;</surname><given-names>S</given-names> </name><name name-style="western"><surname>S&#x00E1;iz</surname><given-names>PA</given-names> </name><name name-style="western"><surname>Bur&#x00F3;n</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Validation of a Spanish version of the Columbia-Suicide Severity Rating Scale (C-SSRS)</article-title><source>Rev Psiquiatr Salud Ment</source><year>2016</year><volume>9</volume><issue>3</issue><fpage>134</fpage><lpage>142</lpage><pub-id pub-id-type="doi">10.1016/j.rpsm.2016.02.002</pub-id><pub-id pub-id-type="medline">27158026</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Radford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Brockman</surname><given-names>G</given-names> </name><name name-style="western"><surname>McLeavey</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sutskever</surname><given-names>I</given-names> </name></person-group><article-title>Robust speech recognition via large-scale weak supervision</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 6, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2212.04356</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><name name-style="western"><surname>Long</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Qwen3 embedding: advancing text embedding and reranking through foundation models</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 11, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2506.05176</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Muennighoff</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tazi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Magne</surname><given-names>L</given-names> </name><name name-style="western"><surname>Reimers</surname><given-names>N</given-names> </name></person-group><article-title>MTEB: massive text embedding benchmark</article-title><conf-name>Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics</conf-name><conf-date>May 2-6, 2023</conf-date><pub-id pub-id-type="doi">10.18653/v1/2023.eacl-main.148</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>McInnes</surname><given-names>L</given-names> </name><name name-style="western"><surname>Healy</surname><given-names>J</given-names> </name><name name-style="western"><surname>Melville</surname><given-names>J</given-names> </name></person-group><article-title>UMAP: Uniform Manifold Approximation and Projection for dimension reduction</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 18, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.1802.03426</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McInnes</surname><given-names>L</given-names> </name><name name-style="western"><surname>Healy</surname><given-names>J</given-names> </name><name name-style="western"><surname>Astels</surname><given-names>S</given-names> </name></person-group><article-title>hdbscan: hierarchical density based clustering</article-title><source>JOSS</source><year>2017</year><volume>2</volume><issue>11</issue><fpage>205</fpage><pub-id pub-id-type="doi">10.21105/joss.00205</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Li</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Qwen3 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  May 14, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.09388</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name></person-group><source>Statistical Power Analysis for the Behavioral Sciences</source><year>1988</year><access-date>2026-07-29</access-date><edition>2</edition><publisher-name>Lawrence Erlbaum Associates</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://utstat.toronto.edu/~brunner/oldclass/378f16/readings/CohenPower.pdf">https://utstat.toronto.edu/~brunner/oldclass/378f16/readings/CohenPower.pdf</ext-link></comment></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fried</surname><given-names>EI</given-names> </name></person-group><article-title>The 52 symptoms of major depression: lack of content overlap among seven common depression scales</article-title><source>J Affect Disord</source><year>2017</year><month>01</month><day>15</day><volume>208</volume><fpage>191</fpage><lpage>197</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2016.10.019</pub-id><pub-id pub-id-type="medline">27792962</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Menne</surname><given-names>F</given-names> </name><name name-style="western"><surname>D&#x00F6;rr</surname><given-names>F</given-names> </name><name name-style="western"><surname>Schr&#x00E4;der</surname><given-names>J</given-names> </name><etal/></person-group><article-title>The voice of depression: speech features as biomarkers for major depressive disorder</article-title><source>BMC Psychiatry</source><year>2024</year><month>11</month><day>12</day><volume>24</volume><issue>1</issue><fpage>794</fpage><pub-id pub-id-type="doi">10.1186/s12888-024-06253-6</pub-id><pub-id pub-id-type="medline">39533239</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hanna</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Wakene</surname><given-names>AD</given-names> </name><name name-style="western"><surname>Johnson</surname><given-names>AO</given-names> </name><name name-style="western"><surname>Lehmann</surname><given-names>CU</given-names> </name><name name-style="western"><surname>Medford</surname><given-names>RJ</given-names> </name></person-group><article-title>Assessing racial and ethnic bias in text generation by large language models for health care-related tasks: cross-sectional study</article-title><source>J Med Internet Res</source><year>2025</year><month>03</month><day>13</day><volume>27</volume><issue>1</issue><fpage>e57257</fpage><pub-id pub-id-type="doi">10.2196/57257</pub-id><pub-id pub-id-type="medline">40080818</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lau</surname><given-names>GR</given-names> </name><name name-style="western"><surname>Low</surname><given-names>WY</given-names> </name><name name-style="western"><surname>Koh</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Nah</surname><given-names>FFH</given-names> </name><name name-style="western"><surname>Hartanto</surname><given-names>A</given-names> </name></person-group><article-title>Evaluating AI alignment in LLMs: output analysis of value priorities across 75 models with human benchmarking</article-title><source>arXiv</source><comment>Preprint posted online on  May 16, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2506.12617</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yasugaki</surname><given-names>S</given-names> </name><name name-style="western"><surname>Okamura</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kaneko</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hayashi</surname><given-names>Y</given-names> </name></person-group><article-title>Bidirectional relationship between sleep and depression</article-title><source>Neurosci Res</source><year>2025</year><month>02</month><volume>211</volume><fpage>57</fpage><lpage>64</lpage><pub-id pub-id-type="doi">10.1016/j.neures.2023.04.006</pub-id><pub-id pub-id-type="medline">37116584</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arena</surname><given-names>AF</given-names> </name><name name-style="western"><surname>Mobbs</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sanatkar</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Mental health and unemployment: a systematic review and meta-analysis of interventions to improve depression and anxiety outcomes</article-title><source>J Affect Disord</source><year>2023</year><month>08</month><day>15</day><volume>335</volume><fpage>450</fpage><lpage>472</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2023.05.027</pub-id><pub-id pub-id-type="medline">37201898</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>P&#x00E9;rez-Jorge</surname><given-names>D</given-names> </name><name name-style="western"><surname>Boutaba-Alehyan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gonz&#x00E1;lez-Contreras</surname><given-names>AI</given-names> </name><name name-style="western"><surname>P&#x00E9;rez-P&#x00E9;rez</surname><given-names>I</given-names> </name></person-group><article-title>Examining the effects of academic stress on student well-being in higher education</article-title><source>Humanit Soc Sci Commun</source><year>2025</year><volume>12</volume><issue>1</issue><fpage>449</fpage><pub-id pub-id-type="doi">10.1057/s41599-025-04698-y</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Fancourt</surname><given-names>D</given-names> </name><name name-style="western"><surname>Finn</surname><given-names>S</given-names> </name></person-group><article-title>What is the evidence on the role of the arts in improving health and well-being? a scoping review</article-title><year>2019</year><access-date>2026-07-29</access-date><publisher-name>WHO Regional Office for Europe</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="http://www.ncbi.nlm.nih.gov/books/NBK553773/">http://www.ncbi.nlm.nih.gov/books/NBK553773/</ext-link></comment></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Soga</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gaston</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Yamaura</surname><given-names>Y</given-names> </name></person-group><article-title>Gardening is beneficial for health: a meta-analysis</article-title><source>Prev Med Rep</source><year>2017</year><month>03</month><volume>5</volume><fpage>92</fpage><lpage>99</lpage><pub-id pub-id-type="doi">10.1016/j.pmedr.2016.11.007</pub-id><pub-id pub-id-type="medline">27981022</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Bloom</surname><given-names>J</given-names> </name><name name-style="western"><surname>Geurts</surname><given-names>SAE</given-names> </name><name name-style="western"><surname>Kompier</surname><given-names>MAJ</given-names> </name></person-group><article-title>Vacation (after-) effects on employee health and well-being, and the role of vacation activities, experiences and sleep</article-title><source>J Happiness Stud</source><year>2013</year><month>04</month><volume>14</volume><issue>2</issue><fpage>613</fpage><lpage>633</lpage><pub-id pub-id-type="doi">10.1007/s10902-012-9345-3</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pearce</surname><given-names>M</given-names> </name><name name-style="western"><surname>Garcia</surname><given-names>L</given-names> </name><name name-style="western"><surname>Abbas</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Association between physical activity and risk of depression: a systematic review and meta-analysis</article-title><source>JAMA Psychiatry</source><year>2022</year><month>06</month><day>1</day><volume>79</volume><issue>6</issue><fpage>550</fpage><lpage>559</lpage><pub-id pub-id-type="doi">10.1001/jamapsychiatry.2022.0609</pub-id><pub-id pub-id-type="medline">35416941</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>H</given-names> </name><name name-style="western"><surname>Phung</surname><given-names>D</given-names> </name><name name-style="western"><surname>Buntine</surname><given-names>W</given-names> </name><name name-style="western"><surname>Du</surname><given-names>L</given-names> </name></person-group><article-title>LLM reading tea leaves: automatically evaluating topic models with large language models</article-title><source>Trans Assoc Comput Linguist</source><year>2025</year><month>04</month><day>17</day><volume>13</volume><fpage>357</fpage><lpage>375</lpage><pub-id pub-id-type="doi">10.1162/tacl_a_00744</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>D&#x2019;Souza</surname><given-names>J</given-names> </name></person-group><article-title>Toward purpose-oriented topic model evaluation enabled by large language models</article-title><source>Int J Digit Libr</source><year>2025</year><month>12</month><volume>26</volume><issue>4</issue><fpage>23</fpage><pub-id pub-id-type="doi">10.1007/s00799-025-00429-5</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Taylor</surname><given-names>B</given-names> </name><name name-style="western"><surname>Henshall</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kenyon</surname><given-names>S</given-names> </name><name name-style="western"><surname>Litchfield</surname><given-names>I</given-names> </name><name name-style="western"><surname>Greenfield</surname><given-names>S</given-names> </name></person-group><article-title>Can rapid approaches to qualitative analysis deliver timely, valid findings to clinical leaders? A mixed methods study comparing rapid and thematic analysis</article-title><source>BMJ Open</source><year>2018</year><month>10</month><day>8</day><volume>8</volume><issue>10</issue><fpage>e019993</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2017-019993</pub-id><pub-id pub-id-type="medline">30297341</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>O&#x2019;Connor</surname><given-names>C</given-names> </name><name name-style="western"><surname>Joffe</surname><given-names>H</given-names> </name></person-group><article-title>Intercoder reliability in qualitative research: debates and practical guidelines</article-title><source>Int J Qual Methods</source><year>2020</year><month>01</month><day>1</day><volume>19</volume><fpage>1609406919899220</fpage><pub-id pub-id-type="doi">10.1177/1609406919899220</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Prescott</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Yeager</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ham</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Comparing the efficacy and efficiency of human and generative AI: qualitative thematic analyses</article-title><source>JMIR AI</source><year>2024</year><month>08</month><day>2</day><volume>3</volume><fpage>e54482</fpage><pub-id pub-id-type="doi">10.2196/54482</pub-id><pub-id pub-id-type="medline">39094113</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bijker</surname><given-names>R</given-names> </name><name name-style="western"><surname>Merkouris</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Dowling</surname><given-names>NA</given-names> </name><name name-style="western"><surname>Rodda</surname><given-names>SN</given-names> </name></person-group><article-title>ChatGPT for automated qualitative research: content analysis</article-title><source>J Med Internet Res</source><year>2024</year><month>07</month><day>25</day><volume>26</volume><fpage>e59050</fpage><pub-id pub-id-type="doi">10.2196/59050</pub-id><pub-id pub-id-type="medline">39052327</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hill</surname><given-names>C</given-names> </name><name name-style="western"><surname>Dahil</surname><given-names>A</given-names> </name><name name-style="western"><surname>Simpson</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Large language models for thematic analysis in healthcare research: a blinded mixed-methods comparison with human analysts</article-title><source>PLOS Digit Health</source><year>2026</year><month>04</month><volume>5</volume><issue>4</issue><fpage>e0001189</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0001189</pub-id><pub-id pub-id-type="medline">41931471</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Castellanos</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Gomes</surname><given-names>P</given-names> </name><name name-style="western"><surname>Vander Meer</surname><given-names>D</given-names> </name><name name-style="western"><surname>Castillo</surname><given-names>A</given-names> </name></person-group><article-title>Large language models for thematic summarization in qualitative health care research: comparative analysis of model and human performance</article-title><source>JMIR AI</source><year>2025</year><month>04</month><day>4</day><volume>4</volume><fpage>e64447</fpage><pub-id pub-id-type="doi">10.2196/64447</pub-id><pub-id pub-id-type="medline">40611510</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Descriptive statistics and sample sizes for the 4 cohorts.</p><media xlink:href="mental_v13i1e99185_app1.pdf" xlink:title="PDF File, 81 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Covariate-adjusted confound analysis for the French general population cohort.</p><media xlink:href="mental_v13i1e99185_app2.pdf" xlink:title="PDF File, 960 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Effect sizes across interview questions and clinical scores for the 4 cohorts.</p><media xlink:href="mental_v13i1e99185_app3.pdf" xlink:title="PDF File, 1077 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Clustering stability and sensitivity across random seeds and minimum cluster size, internal cluster-validity indices, and bootstrap CIs for the reported effect sizes.</p><media xlink:href="mental_v13i1e99185_app4.pdf" xlink:title="PDF File, 217 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Pipeline hyperparameters and the cluster-description prompt.</p><media xlink:href="mental_v13i1e99185_app5.pdf" xlink:title="PDF File, 102 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6</label><p>Stability of the language-model cluster descriptions across transcript sampling, generation seeds, and prompt wording.</p><media xlink:href="mental_v13i1e99185_app6.pdf" xlink:title="PDF File, 166 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7</label><p>Small-cohort Monte-Carlo permutation tests with expected cell counts.</p><media xlink:href="mental_v13i1e99185_app7.pdf" xlink:title="PDF File, 120 KB"/></supplementary-material><supplementary-material id="app8"><label>Multimedia Appendix 8</label><p>Keyword labels vs language-model descriptions for the same clusters.</p><media xlink:href="mental_v13i1e99185_app8.pdf" xlink:title="PDF File, 167 KB"/></supplementary-material><supplementary-material id="app9"><label>Multimedia Appendix 9</label><p>Faithfulness of the cluster descriptions to their transcripts.</p><media xlink:href="mental_v13i1e99185_app9.pdf" xlink:title="PDF File, 100 KB"/></supplementary-material><supplementary-material id="app10"><label>Multimedia Appendix 10</label><p>Computational cost and comparison with manual thematic analysis.</p><media xlink:href="mental_v13i1e99185_app10.pdf" xlink:title="PDF File, 117 KB"/></supplementary-material></app-group></back></article>