<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.4 20241031//EN" "JATS-journalpublishing1-4.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.4" xml:lang="en">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">Oalib</journal-id>
      <journal-title-group>
        <journal-title>Open Access Library Journal</journal-title>
      </journal-title-group>
      <issn pub-type="epub">2333-9721</issn>
      <issn pub-type="ppub">2333-9705</issn>
      <publisher>
        <publisher-name>Scientific Research Publishing</publisher-name>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.4236/oalib.1115676</article-id>
      <article-id pub-id-type="publisher-id">Oalib-154126</article-id>
      <article-categories>
        <subj-group>
          <subject>Article</subject>
        </subj-group>
        <subj-group>
          <subject>Biomedical</subject>
          <subject>Life Sciences</subject>
          <subject>Business</subject>
          <subject>Economics</subject>
          <subject>Chemistry</subject>
          <subject>Materials Science</subject>
          <subject>Computer Science</subject>
          <subject>Communications</subject>
          <subject>Earth</subject>
          <subject>Environmental Sciences</subject>
          <subject>Engineering</subject>
          <subject>Medicine</subject>
          <subject>Healthcare</subject>
          <subject>Physics</subject>
          <subject>Mathematics</subject>
          <subject>Social Sciences</subject>
          <subject>Humanities</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Simulation-Based Natural Language Processing of Synthetic Clinical Notes to Detect Early Linguistic Markers of Antidepressant-Induced Hypomania in Bipolar Disorder: A Proof-of-Concept Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author" corresp="yes">
          <contrib-id contrib-id-type="orcid">0000-0001-9101-072X</contrib-id>
          <name name-style="western">
            <surname>Filippis</surname>
            <given-names>Rocco de</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <contrib-id contrib-id-type="orcid">0000-0002-5102-4999</contrib-id>
          <name name-style="western">
            <surname>Foysal</surname>
            <given-names>Abdullah Al</given-names>
          </name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
      </contrib-group>
      <aff id="aff1"><label>1</label> Institute of Psychopathology, Rome, Italy </aff>
      <aff id="aff2"><label>2</label> ARAVIS Unit, EPSM 74, La Roche-sur-Foron, Haute-Savoie, France </aff>
      <aff id="aff3"><label>3</label> Department of Informatics, Bioengineering, Robotics and Systems Engineering, University of Genoa, Genoa, Italy </aff>
      <author-notes>
        <fn fn-type="conflict" id="fn-conflict">
          <p>The authors declare no conflicts of interest.</p>
        </fn>
      </author-notes>
      <pub-date pub-type="epub">
        <day>02</day>
        <month>09</month>
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="collection">
        <month>09</month>
        <year>2026</year>
      </pub-date>
      <volume>13</volume>
      <issue>09</issue>
      <fpage>1</fpage>
      <lpage>16</lpage>
      <history>
        <date date-type="received">
          <day>22</day>
          <month>06</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>20</day>
          <month>09</month>
          <year>2026</year>
        </date>
        <date date-type="published">
          <day>23</day>
          <month>09</month>
          <year>2026</year>
        </date>
      </history>
      <permissions>
        <copyright-statement>© 2026 by the authors and Scientific Research Publishing Inc.</copyright-statement>
        <copyright-year>2026</copyright-year>
        <license license-type="open-access">
          <license-p> This article is an open access article distributed under the terms and conditions of the Creative Commons Attribution (CC BY) license ( <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link> ). </license-p>
        </license>
      </permissions>
      <self-uri content-type="doi" xlink:href="https://doi.org/10.4236/oalib.1115676">https://doi.org/10.4236/oalib.1115676</self-uri>
      <abstract>
        <p>Antidepressant-induced hypomania in bipolar disorder (BD) is a clinically consequential pharmacological safety event that may be preceded by changes in patient and clinician language. This proof-of-concept simulation study evaluates whether natural language processing (NLP) can recover such signals from fully synthetic outpatient clinical notes and simulated hypomania labels; it does not establish clinical validity. We generated one synthetic note for each of N = 700 simulated antidepressant-exposed BD patients (hypomania prevalence 25.1%). A three-tier feature architecture encoded: i) 18 psycholinguistic and syntactic features; ii) 30 latent semantic analysis components derived from TF-IDF representations; and iii) 12 simulated clinical meta-features. A four-layer multilayer perceptron, termed ClinicalBERT-sim to distinguish it from an actually fine-tuned ClinicalBERT model, was trained on the combined feature matrix. An out-of-fold stacking ensemble fused ClinicalBERT-sim, LightGBM, and Random Forest outputs. The proposed ensemble achieved AUC = 0.912 (95% CI: 0.842 - 0.969), F1 = 0.708, sensitivity = 0.595, and specificity = 0.974 on the held-out synthetic test set. SHAP analysis identified hypomanic term score, word count, energy word score, physician concern score, and positive affect ratio as the dominant simulated predictors. Longitudinal illustrative trajectories showed rising hypomanic term score and word count across simulated weekly notes before simulated onset. These findings demonstrate technical feasibility within the designed simulator, but performance may partly reflect the rules used to generate notes and labels. Validation on independently annotated, real outpatient BD notes is required before any clinical screening or workflow use.</p>
      </abstract>
      <kwd-group kwd-group-type="author-generated" xml:lang="en">
        <kwd>Natural Language Processing</kwd>
        <kwd>Clinical Notes</kwd>
        <kwd>Bipolar Disorder</kwd>
        <kwd>Antidepressant-Induced Hypomania</kwd>
        <kwd>Psycholinguistics</kwd>
        <kwd>TF-IDF</kwd>
        <kwd>Latent Semantic Analysis</kwd>
        <kwd>SHAP</kwd>
        <kwd>Linguistic Biomarkers</kwd>
        <kwd>Digital Phenotyping</kwd>
        <kwd>EHR</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec1">
      <title>1. Introduction</title>
      <p>Psychiatric clinical notes are among the richest and most systematically underutilised data sources in mental health surveillance. In the outpatient management of bipolar disorder with antidepressant exposure, clinicians generate free-text progress notes at every visit that contain not only structured observations (YMRS scores, medication changes) but also nuanced language that reflects the patient’s mental state in ways that structured rating scales cannot fully capture: the word choice, sentence structure, and affect conveyed in how a patient describes their week, their energy, their sleep, and their plans are direct signals of their phenomenological state [<xref ref-type="bibr" rid="B1">1</xref>].</p>
      <p>The hypomanic prodrome in antidepressant-exposed BD patients is characterised by a predictable sequence of linguistic changes that begin before the clinical threshold for formal hypomania is reached [<xref ref-type="bibr" rid="B2">2</xref>]. Patients speak and write with more words, more energetic vocabulary, more positive affect language, and more future-oriented statements. Physicians, sensitive to these changes, begin expressing concern in their notes: the lexical field of clinical notes shifts from maintenance language to monitoring language [<xref ref-type="bibr" rid="B3">3</xref>]. These changes are legible to a trained psychiatrist who reads serial notes; the questions this paper addresses are whether they are also detectable by NLP models trained on these notes, and whether they are detectable early enough to be clinically actionable i.e., before the full hypomanic episode emerges.</p>
      <p>The NLP evidence base for mental health applications has expanded with domain-specific language models, including BioBERT [<xref ref-type="bibr" rid="B4">4</xref>], ClinicalBERT [<xref ref-type="bibr" rid="B5">5</xref>], and PsychBERT [<xref ref-type="bibr" rid="B6">6</xref>]. However, PsychBERT was developed for social-media mental-health analysis rather than clinical notes, and the present study does not fine-tune any BERT model. Instead, it uses an MLP on engineered linguistic, LSA, and clinical features as a simulation proxy, labelled ClinicalBERT-sim throughout to avoid implying use of contextual ClinicalBERT embeddings.</p>
      <p>We present five proof-of-concept contributions in a fully synthetic setting: i) a simulation framework for studying antidepressant-induced hypomania signals in outpatient-style BD notes; ii) a three-tier feature architecture integrating psycholinguistic/syntactic features, LSA embeddings, and simulated clinical meta-features; iii) comparison of a deep multilayer perceptron proxy with conventional text and tree-based baselines; iv) four linguistic visualisations; and v) subgroup analyses within the simulated cohort. These contributions concern methodological feasibility and should not be interpreted as evidence of clinical prediction performance.</p>
    </sec>
    <sec id="sec2">
      <title>2. Background and Related Work</title>
      <sec id="sec2dot1">
        <title>2.1. Linguistic Markers of Hypomania and Mania</title>
        <p>The relationship between language and affective state in bipolar disorder has been documented since Kraepelin’s clinical descriptions of pressured speech and flight of ideas as hallmarks of manic activation [<xref ref-type="bibr" rid="B7">7</xref>]. Quantitative linguistic analysis has confirmed that hypomanic and manic states are associated with increased word production, reduced lexical diversity (type-token ratio), elevated positive affect word density, increased self-referential pronouns, and a higher proportion of future-oriented language [<xref ref-type="bibr" rid="B8">8</xref>]. Automated speech analysis has detected hypomanic language patterns with high sensitivity using prosodic features (speech rate, fundamental frequency variability) and lexical features (energy vocabulary, social language) [<xref ref-type="bibr" rid="B9">9</xref>].</p>
      </sec>
      <sec id="sec2dot2">
        <title>2.2. NLP of Clinical Notes in Psychiatry</title>
        <p>Recent peer-reviewed literature supports the broader use of NLP in psychiatric and clinical text, but not the specific real-world applications previously attributed here. Reviews have mapped NLP applications across neuroscience and psychiatry [<xref ref-type="bibr" rid="B10">10</xref>] and specifically within bipolar-disorder research [<xref ref-type="bibr" rid="B11">11</xref>]. A 2024 study evaluated classification of unstructured EHR text by psychiatric diagnosis [<xref ref-type="bibr" rid="B12">12</xref>], while a 2025 bipolar-disorder pilot combined acoustic and natural-language markers for remote symptom assessment [<xref ref-type="bibr" rid="B13">13</xref>]. Together, these studies support the feasibility of psychiatric language analysis while also highlighting limited labelled data, domain shift, and the need for external clinical validation.</p>
      </sec>
      <sec id="sec2dot3">
        <title>2.3. Clinical Language Models</title>
        <p>ClinicalBERT is a BERT model adapted to de-identified clinical notes, whereas BioBERT was pretrained primarily on biomedical literature. PsychBERT was developed for mental-health language in social media. Recent reviews emphasise that clinical NLP models require task-specific labelled data, transparent validation, and careful treatment of bias and generalisability. Because no labelled corpus of antidepressant-induced hypomania notes was available for this study, we did not fine-tune ClinicalBERT or PsychBERT; the term ClinicalBERT-sim denotes only a multilayer perceptron operating on engineered and LSA features, not a transformer language model.</p>
      </sec>
      <sec id="sec2dot4">
        <title>2.4. Psycholinguistic Feature Frameworks</title>
        <p>The Linguistic Inquiry and Word Count (LIWC) framework [<xref ref-type="bibr" rid="B14">14</xref>] provides validated dictionaries for psycholinguistic categories including positive and negative affect, social language, cognition, biological processes, and temporal orientation. LIWC-derived features have shown strong performance in detecting mental health conditions from text, including depression (r = 0.32 with BDI scores) and anxiety (positive affect ratio inversely correlated with GAD-7) [<xref ref-type="bibr" rid="B15">15</xref>][<xref ref-type="bibr" rid="B16">16</xref>]. Our feature extraction framework is inspired by but not directly dependent on LIWC, deriving analogous category scores from domain-specific hypomanic lexicons constructed from clinical and empirical sources.</p>
      </sec>
    </sec>
    <sec id="sec3">
      <title>3. Methods</title>
      <sec id="sec3dot1">
        <title>3.1. Dataset Design</title>
        <p>This proof-of-concept study used a fully synthetic dataset of N = 700 simulated outpatient BD patients receiving antidepressant-containing regimens (BD-I: 55%, BD-II: 45%). Each simulated patient contributed one synthetic note at a 4-week post-initiation review and one simulated binary label indicating hypomania emergence during the subsequent 4 weeks. No real EHR text, patient identifiers, or clinical outcomes were used [<xref ref-type="bibr" rid="B17">17</xref>][<xref ref-type="bibr" rid="B18">18</xref>].</p>
      </sec>
      <sec id="sec3dot2">
        <title>3.2. Synthetic Clinical Note Generation</title>
        <p>Synthetic notes were generated with a controlled template-and-lexicon procedure. Note length, sentence count, lexical diversity, and punctuation were sampled from bounded distributions intended to resemble outpatient progress-note structure. Hypomanic lexical pools covered energy, reduced sleep need, elevated mood, pressured speech, goal-directed activity, future orientation, and grandiosity; stable/depressive pools covered mood stability, adequate sleep, medication continuation, and residual depressive symptoms. A latent simulated risk score combined BD subtype, antidepressant-class risk, mood-stabiliser adequacy, baseline YMRS and MADRS, interdaily stability, genetic-risk composite, prior antidepressant-associated hypomania, prior antidepressant trials, illness duration, age, and sex. The binary hypomania label was sampled from a logistic transformation of this risk score with added stochastic noise, and note-generation probabilities were conditioned on the same latent score. Subgroups were assigned directly from the corresponding simulated meta-features. Consequently, linguistic markers and clinical meta-features were not independent of label generation, and the resulting performance may partly represent recovery of simulator rules rather than discovery of clinical biomarkers. Physician concern language was sampled with a probability that increased with latent risk, making it an intentionally informative but potentially advantaged feature.</p>
      </sec>
      <sec id="sec3dot3">
        <title>3.3. Feature Architecture</title>
        <p>Psycholinguistic and syntactic features (18) comprised hypomanic term score, positive and negative affect ratios, physician concern score, word count, sentence count, mean and standard deviation of sentence length, type-token ratio, exclamation rate, first-person pronoun rate, future-orientation score, energy-word score, sleep-word score, note length, number density, punctuation density, and upper-case letter rate. LSA features (30) were derived from TF-IDF representations using a maximum vocabulary of 500 terms, unigrams and bigrams, minimum document frequency of 3, and truncated SVD. The 12 simulated clinical meta-features were encoded as follows: BD subtype and sex as binary indicators; antidepressant-class risk as a prespecified ordinal score; mood-stabiliser adequacy as a continuous 0 - 1 score combining therapeutic dose/level and adherence; baseline YMRS and MADRS as continuous scale values at note generation; interdaily stability as a continuous 0 - 1 circadian-regularity index; genetic risk as a standardised synthetic polygenic/family-history composite; prior antidepressant-associated hypomania as binary; illness duration, prior antidepressant trials, and age as continuous or count variables. These operational definitions apply only to the simulator.</p>
      </sec>
      <sec id="sec3dot4">
        <title>3.4. Model Architecture</title>
        <p>Five NLP models were evaluated. TF-IDF + LR used L2-regularised logistic regression (C = 0.1) on TF-IDF unigrams and bigrams. Linguistic XGBoost used the 18 psycholinguistic/syntactic features only. ClinicalBERT-sim was a four-layer multilayer perceptron (D -&gt; 256 -&gt; 128 -&gt; 64 -&gt; 2) trained on the combined engineered linguistic, LSA, and simulated clinical feature matrix; despite its historical name in the analysis code, it is not ClinicalBERT and does not use transformer embeddings. Random Forest and LightGBM used the same combined feature matrix. The proposed ensemble combined out-of-fold predictions from ClinicalBERT-sim, LightGBM, and Random Forest through a logistic-regression meta-learner.</p>
      </sec>
      <sec id="sec3dot5">
        <title>3.5. Evaluation Framework</title>
        <p>The 700 patient-level records were stratified by simulated outcome and split once into training (n = 490; 70%), validation (n = 105; 15%), and test (n = 105; 15%) partitions using master random seed 42. Each patient contributed exactly one note, so no patient or overlapping text window appeared in more than one partition. The TF-IDF vocabulary, inverse-document-frequency weights, truncated-SVD projection, scaling parameters, lexicon-derived normalisation, model fitting, and out-of-fold stacking were fitted using training data only. Hyperparameter and early-stopping decisions used the validation set. The classification threshold of 0.53 was selected on the validation set by maximising F1 and then locked before the test set was evaluated. Evaluation included AUC, F1, sensitivity, specificity with 95% bootstrap confidence intervals (1000 resamples), Brier score, decision-curve analysis, Bayesian model comparison, SHAP analysis for the linguistic XGBoost model, TF-IDF coefficient analysis, and prespecified simulated subgroup analyses.</p>
      </sec>
      <sec id="sec3dot6">
        <title>3.6. Reproducibility</title>
        <p>The hypomania lexicon was assembled from terms describing elevated or irritable mood, increased energy, reduced need for sleep, pressured speech, flight of ideas, goal-directed activity, future orientation, and grandiosity; stable-note terms represented euthymia, adequate sleep, medication continuation, and residual depression [<xref ref-type="bibr" rid="B19">19</xref>]. Generation parameters included target note-length ranges, sentence-count ranges, class-conditional term-insertion probabilities, physician-concern probabilities, label-noise probability, and the coefficients of the simulated risk function. The master data-generation seed was 42, and model initialisation seeds were 101, 202, and 303 for ensemble components. For publication, the authors should deposit the exact generation script, lexicons, prompts/templates, configuration file, and analysis code in a versioned repository; until these materials are released, the study is not fully independently reproducible.</p>
      </sec>
      <sec id="sec3dot7">
        <title>3.7. Ethics and Data Governance</title>
        <p>No real patient notes, protected health information, or identifiable clinical records were accessed or processed. All notes, meta-features, subgroup assignments, and outcome labels were generated synthetically. On that basis, institutional review board approval and informed consent were not required for this computational simulation study. This statement should be confirmed against the authors’ institutional policies before submission. Any future study using real psychiatric notes will require formal ethics review, data-processing agreements, secure access controls, de-identification, and governance procedures appropriate to highly sensitive mental-health information.</p>
      </sec>
    </sec>
    <sec id="sec4">
      <title>4. Results</title>
      <sec id="sec4dot1">
        <title>4.1. Calibration and Clinical Utility</title>
        <p><xref ref-type="fig" rid="fig1">Figure 1</xref><xref ref-type="fig" rid="fig1">Figure 1</xref> presents calibration curves in the synthetic test set. The proposed ensemble achieved Brier score = 0.096, while LightGBM achieved the numerically lowest Brier score (0.090). <xref ref-type="fig" rid="fig2">Figure 2</xref><xref ref-type="fig" rid="fig2">Figure 2</xref> presents decision-curve analysis within the simulated data. These analyses describe utility under the simulator assumptions only and do not establish clinical benefit or justify changes to antidepressant treatment.</p>
      </sec>
      <sec id="sec4dot2">
        <title>4.2. Discriminative Performance</title>
        <p><bold>Table 1</bold> presents comparative performance on the held-out synthetic test set (n = 105). The proposed ensemble achieved AUC = 0.912 (95% CI: 0.842 - 0.969), F1 = 0.708, sensitivity = 0.595, and specificity = 0.974. LightGBM achieved a slightly higher AUC (0.916) and F1 (0.720), whereas the ensemble had higher specificity. ClinicalBERT-sim achieved AUC = 0.907 versus TF-IDF + LR at 0.791. Because ClinicalBERT-sim is an MLP on engineered and LSA features rather than a fine-tuned transformer, this comparison should not be interpreted as evidence that contextual ClinicalBERT representations outperform TF-IDF. All results are conditional on synthetic note and label generation.</p>
        <fig id="fig1">
          <label>Figure 1</label>
          <graphic xlink:href="https://html.scirp.org/file/1115676-rId16.jpeg?20260923020530" />
        </fig>
        <p><bold>Figure 1.</bold>Calibration curves for all six models. Brier scores: Proposed Ensemble 0.096, ClinBERT Sim. 0.107, LightGBM 0.090, Linguistic XGB 0.134, Random Forest 0.107, TF-IDF + LR 0.162. Perfect calibration = dashed diagonal.</p>
        <fig id="fig2">
          <label>Figure 2</label>
          <graphic xlink:href="https://html.scirp.org/file/1115676-rId17.jpeg?20260923020530" />
        </fig>
        <p><bold>Figure 2.</bold>Decision curve analysis. Proposed Ensemble (dark bold) achieves highest net benefit across threshold probabilities 0.20 - 0.50. High specificity profile limits false-positive antidepressant interruptions in the stable majority.</p>
        <p><bold>Table 1.</bold>Comparative model performance-test set (n = 105).</p>
        <table-wrap id="tbl1">
          <label>Table 1</label>
          <table>
            <tbody>
              <tr>
                <td>
                  <bold>Model</bold>
                </td>
                <td>
                  <bold>AUC</bold>
                </td>
                <td>
                  <bold>F1</bold>
                </td>
                <td>
                  <bold>Sensitivity</bold>
                </td>
                <td>
                  <bold>Specificity</bold>
                </td>
                <td>
                  <bold>Brier</bold>
                </td>
              </tr>
              <tr>
                <td>TF-IDF + LR</td>
                <td>0.791</td>
                <td>0.629</td>
                <td>0.670</td>
                <td>0.844</td>
                <td>0.162</td>
              </tr>
              <tr>
                <td>Linguistic XGBoost</td>
                <td>0.796</td>
                <td>0.666</td>
                <td>0.632</td>
                <td>0.911</td>
                <td>0.134</td>
              </tr>
              <tr>
                <td>ClinBERT Simulation</td>
                <td>0.907</td>
                <td>0.734</td>
                <td>0.629</td>
                <td>0.974</td>
                <td>0.107</td>
              </tr>
              <tr>
                <td>Random Forest</td>
                <td>0.890</td>
                <td>0.708</td>
                <td>0.595</td>
                <td>0.974</td>
                <td>0.107</td>
              </tr>
              <tr>
                <td>LightGBM</td>
                <td>0.916</td>
                <td>0.720</td>
                <td>0.672</td>
                <td>0.936</td>
                <td>0.090</td>
              </tr>
              <tr>
                <td>
                  <bold>Proposed Ensemble (proposed)</bold>
                </td>
                <td>
                  <bold>0.912</bold>
                </td>
                <td>
                  <bold>0.708</bold>
                </td>
                <td>
                  <bold>0.595</bold>
                </td>
                <td>
                  <bold>0.974</bold>
                </td>
                <td>
                  <bold>0.096</bold>
                </td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>95% bootstrap CI (1000 resamples): AUC [0.842 - 0.969], F1 [0.545 - 0.847], Sens [0.400 - 0.778], Spec [0.934 - 1.000]. Threshold = 0.53, selected on the validation set by maximising F1 and locked before test evaluation. ClinBERT Sim = four-layer MLP on engineered linguistic, LSA, and simulated clinical features; it is not a fine-tuned ClinicalBERT model.</p>
      </sec>
      <sec id="sec4dot3">
        <title>4.3. Linguistic Marker Bubble Chart</title>
        <p><xref ref-type="fig" rid="fig3">Figure 3</xref><xref ref-type="fig" rid="fig3">Figure 3</xref> presents the linguistic marker bubble chart, plotting each psycholinguistic feature’s mean value in hypomanic notes (x-axis) against its SHAP predictive importance (y-axis), with bubble size encoding signed SHAP magnitude and colour encoding risk direction. Three features occupy the high-prevalence, high-SHAP quadrant: hypomanic term score (most prevalent in hypomanic notes and highest SHAP contributor), energy word score (elevated in hypomanic notes, </p>
        <fig id="fig3">
          <label>Figure 3</label>
          <graphic xlink:href="https://html.scirp.org/file/1115676-rId18.jpeg?20260923020530" />
        </fig>
        <p><bold>Figure 3.</bold>Linguistic marker bubble chart. x-axis: mean feature value in hypomanic notes (standardised). y-axis: mean |SHAP| importance. Bubble size: signed SHAP magnitude. Red: risk-increasing features. Blue: risk-reducing features. Hypomanic term score, energy word score, and physician concern score occupy the high-prevalence/high-SHAP quadrant.</p>
        <p>strong positive SHAP), and physician concern score (most discriminating note-level feature clinicians explicitly expressing monitoring language are effectively pre-annotating risk). The type-token ratio (lexical diversity) occupies a distinctive position: it has moderate SHAP importance in the protective direction, consistent with the established finding that hypomanic speech reduces lexical diversity despite increasing word count the model is detecting not just content but structural speech characteristics.</p>
      </sec>
      <sec id="sec4dot4">
        <title>4.4. ROC Curves</title>
        <p><xref ref-type="fig" rid="fig4">Figure 4</xref><xref ref-type="fig" rid="fig4">Figure 4</xref> presents ROC curves with 95% bootstrap confidence bands. ClinicalBERT-sim achieved AUC = 0.907 and TF-IDF + LR achieved AUC = 0.791 (difference 0.116) in this synthetic dataset. The difference reflects the combined engineered, LSA, and simulated clinical inputs available to ClinicalBERT-sim, not a direct comparison between an actual transformer language model and TF-IDF. LightGBM achieved the highest point-estimate AUC (0.916), followed by the ensemble (0.912).</p>
        <fig id="fig4">
          <label>Figure 4</label>
          <graphic xlink:href="https://html.scirp.org/file/1115676-rId19.jpeg?20260923020530" />
        </fig>
        <p><bold>Figure 4.</bold>ROC curves with 95% bootstrap CI bands (400 resamples) in the synthetic test set. ClinicalBERT-sim (AUC = 0.907) uses an MLP on engineered, LSA, and simulated clinical features; it is not a fine-tuned ClinicalBERT model. LightGBM achieved AUC = 0.916 and the proposed ensemble AUC = 0.912.</p>
      </sec>
      <sec id="sec4dot5">
        <title>4.5. SHAP Signed Comparison: Hypomanic vs Stable Notes</title>
        <p><xref ref-type="fig" rid="fig5">Figure 5</xref><xref ref-type="fig" rid="fig5">Figure 5</xref> presents the signed SHAP dual comparison the primary novel visualisation of this paper’s linguistic analysis. The left panel shows the top 12 linguistic predictors in notes that preceded hypomania, with their signed mean SHAP contributions; the right panel shows the same features in notes from stable patients. The divergence between panels reveals not just which features are important but why they differ between groups. In hypomanic notes, word count and hypomanic term score carry strong positive SHAP contributions (the model is detecting the verbose, energy-laden language of the hypomanic prodrome). The physician concern score shows a particularly large positive contribution in hypomanic notes and near-zero in stable notes confirming that the prescribing clinician’s own linguistic alarm signal is the most reliable single marker of emerging hypomania. In stable notes, the note length and pos_affect_ratio are flat or mildly negative, consistent with stable or mildly improving depression producing measured, low-affect clinical documentation.</p>
        <fig id="fig5">
          <label>Figure 5</label>
          <graphic xlink:href="https://html.scirp.org/file/1115676-rId20.jpeg?20260923020530" />
        </fig>
        <p><bold>Figure 5.</bold>Signed SHAP dual comparison—linguistic features in hypomanic (left) vs stable (right) clinical notes. Positive (red) bars increase hypomania risk prediction; negative (blue) reduce it. Key divergences: hypomanic notes show large positive word count and hypomanic term SHAP; physician concern score is the highest positive contributor in hypomanic notes and near-zero in stable notes. Stable notes show uniformly low, flat SHAP magnitudes.</p>
      </sec>
      <sec id="sec4dot6">
        <title>4.6. Longitudinal Linguistic Trajectory</title>
        <p><xref ref-type="fig" rid="fig6">Figure 6</xref><xref ref-type="fig" rid="fig6">Figure 6</xref> presents the longitudinal note trajectory simulated weekly clinical note feature evolution for two hypomanic patients (top panel) and two stable patients (bottom panel). In hypomanic patients, word count, hypomanic term score, positive affect ratio, and physician concern score all show monotonically rising trajectories across 4 - 5 weeks, with the rate of rise accelerating in weeks 3 - 4 before the clinical hypomania threshold is reached. The physician concern score, notably, begins rising at week 3 - 4 after the other linguistic features but before formal clinical diagnosis suggesting that prescriber intuition is activated by the accumulating linguistic signal approximately one week before the patient crosses the formal YMRS threshold. In stable patients, all four features remain flat and low-variance throughout the 6-week monitoring window, with no directional trend.</p>
        <fig id="fig6">
          <label>Figure 6</label>
          <graphic xlink:href="https://html.scirp.org/file/1115676-rId21.jpeg?20260923020530" />
        </fig>
        <p><bold>Figure 6.</bold>Longitudinal linguistic feature trajectories. Top panel: two hypomanic patients showing rising word count (blue), hypomanic term score (red), positive affect ratio (green), and physician concern score (orange) across weekly visits. Vertical dotted lines: hypomania clinical onset (week 4 and 5). Bottom panel: two stable patients showing flat, low-variance profiles across all 6 weeks.</p>
      </sec>
      <sec id="sec4dot7">
        <title>4.7. Bayesian Model Comparison</title>
        <p><bold>Table</bold><bold>2</bold> and <xref ref-type="fig" rid="fig7">Figure 7</xref><xref ref-type="fig" rid="fig7">Figure 7</xref> present the Bayesian comparison. The proposed ensemble achieved BIC = 90.3, the lowest of all models. All competitors exceeded the decisive evidence threshold (log<sub>10</sub>BF &gt; 2) in favour of the ensemble. The BF comparison against ClinBERT simulation (log<sub>10</sub>BF = 44.9) confirms that the meta-learner’s fusion of ClinBERT, LightGBM, and RF outputs adds decisive model-theoretic value beyond any single component, despite ClinBERT simulation already achieving near-ensemble AUC as a standalone model.</p>
        <p><bold>Table 2.</bold>Bayesian model comparison: BIC, WAIC, and Bayes factors.</p>
        <table-wrap id="tbl2">
          <label>Table 2</label>
          <table>
            <tbody>
              <tr>
                <td>
                  <bold>Model</bold>
                </td>
                <td>
                  <bold>Log-Lik.</bold>
                </td>
                <td>
                  <bold>k</bold>
                </td>
                <td>
                  <bold>BIC</bold>
                </td>
                <td>
                  <bold>WAIC</bold>
                </td>
                <td>
                  <bold>log</bold>
                  <bold>
                    <sub>10</sub>
                  </bold>
                  <bold>(BF)</bold>
                </td>
                <td>
                  <bold>Evidence</bold>
                </td>
              </tr>
              <tr>
                <td>TF-IDF + LR</td>
                <td>−77.1</td>
                <td>30</td>
                <td>293.8</td>
                <td>160.9</td>
                <td>44.2</td>
                <td>Decisive</td>
              </tr>
              <tr>
                <td>Linguistic XGBoost</td>
                <td>−53.0</td>
                <td>40</td>
                <td>292.2</td>
                <td>107.6</td>
                <td>43.8</td>
                <td>Decisive</td>
              </tr>
              <tr>
                <td>ClinBERT Simulation</td>
                <td>−43.8</td>
                <td>45</td>
                <td>297.0</td>
                <td>88.7</td>
                <td>44.9</td>
                <td>Decisive</td>
              </tr>
              <tr>
                <td>Random Forest</td>
                <td>−39.5</td>
                <td>40</td>
                <td>265.2</td>
                <td>79.6</td>
                <td>38.0</td>
                <td>Decisive</td>
              </tr>
              <tr>
                <td>LightGBM</td>
                <td>−34.0</td>
                <td>60</td>
                <td>347.3</td>
                <td>69.1</td>
                <td>55.8</td>
                <td>Decisive</td>
              </tr>
              <tr>
                <td>
                  <bold>Proposed Ensemble (proposed)</bold>
                </td>
                <td>
                  <bold>−</bold>
                  <bold>35.9</bold>
                </td>
                <td>
                  <bold>4</bold>
                </td>
                <td>
                  <bold>90.3</bold>
                </td>
                <td>
                  <bold>72.6</bold>
                </td>
                <td>
                  <bold>0.0 (ref.)</bold>
                </td>
                <td>
                  <bold>Reference</bold>
                </td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>k: parameter count. log₁₀(BF): Bayes Factor in favour of Proposed Ensemble. Decisive: log₁₀(BF) &gt; 2.</p>
        <fig id="fig7">
          <label>Figure 7</label>
          <graphic xlink:href="https://html.scirp.org/file/1115676-rId22.jpeg?20260923020530" />
        </fig>
        <p><bold>Figure 7.</bold>Bayesian model comparison. Left: BIC values (Proposed Ensemble = 90.3, lowest). Right: log<sub>10</sub> Bayes Factor evidence; all competitors exceed the decisive threshold (log<sub>10</sub>BF &gt; 2).</p>
      </sec>
      <sec id="sec4dot8">
        <title>4.8. Subgroup Analysis</title>
        <p><bold>Table</bold><bold>3</bold> presents subgroup performance. The elevated baseline YMRS subgroup (YMRS &gt; 6 at antidepressant initiation) achieved the highest subgroup AUC (0.981), consistent with subthreshold manic activation at baseline making the transition to hypomania more linguistically stereotyped and therefore more predictable. High-risk antidepressant class patients achieved AUC = 0.892, confirming that pharmacological class risk amplifies the linguistic signal in ways the model detects. Prior hypomania history patients showed moderate AUC (0.695), reflecting the complex interaction between prior experience and current linguistic expression.</p>
        <p><bold>Table 3.</bold> Subgroup analysis-proposed ensemble performance.</p>
        <table-wrap id="tbl3">
          <label>Table 3</label>
          <table>
            <tbody>
              <tr>
                <td>
                  <bold>Subgroup</bold>
                </td>
                <td>
                  <bold>N</bold>
                </td>
                <td>
                  <bold>AUC</bold>
                </td>
                <td>
                  <bold>F1</bold>
                </td>
                <td>
                  <bold>Sensitivity</bold>
                </td>
              </tr>
              <tr>
                <td>Prior antidepressant hypomania</td>
                <td>24</td>
                <td>0.695</td>
                <td>0.865</td>
                <td>0.842</td>
              </tr>
              <tr>
                <td>BD-I subtype</td>
                <td>56</td>
                <td>0.832</td>
                <td>0.621</td>
                <td>0.500</td>
              </tr>
              <tr>
                <td>No adequate mood stabiliser</td>
                <td>44</td>
                <td>0.845</td>
                <td>0.609</td>
                <td>0.500</td>
              </tr>
              <tr>
                <td>High-risk AD class</td>
                <td>23</td>
                <td>0.892</td>
                <td>0.500</td>
                <td>0.333</td>
              </tr>
              <tr>
                <td>Elevated baseline YMRS (&gt;6)</td>
                <td>30</td>
                <td>0.981</td>
                <td>0.737</td>
                <td>0.583</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>Elevated baseline YMRS: YMRS &gt; 6 at antidepressant initiation. High-risk AD class: pharmacological risk index &gt; 0.28. All subgroup n ≥ 12.</p>
      </sec>
      <sec id="sec4dot9">
        <title>4.9. Discriminative Terms Analysis</title>
        <p><xref ref-type="fig" rid="fig8">Figure 8</xref><xref ref-type="fig" rid="fig8">Figure 8</xref> presents the top 15 discriminative n-grams from the TF-IDF + LR model, shown as diverging horizontal bars (hypomanic terms left, stable/depressive terms right). The hypomanic term list reflects the hypomanic prodrome lexicon: energy vocabulary, elevated mood descriptors, reduced sleep language, and physician activation language. The stable/depressive term list reflects maintenance psychiatry language: stable mood descriptors, adequate sleep, continuing medication, and mild residual depressive terms. The term-level analysis provides the most directly interpretable NLP output for clinical deployment: specific words and bigrams that should trigger heightened monitoring when identified in a clinical note.</p>
        <fig id="fig8">
          <label>Figure 8</label>
          <graphic xlink:href="https://html.scirp.org/file/1115676-rId23.jpeg?20260923020530" />
        </fig>
        <p><bold>Figure 8.</bold>Discriminative clinical note terms. Left panel: top 15 hypomanic-associated n-grams (highest TF-IDF + LR positive coefficients). Right panel: top 15 stable/depressive-associated n-grams (most negative coefficients). Bar length = coefficient magnitude. Terms directly reflect the clinical hypomanic prodrome lexicon and stable maintenance psychiatry language.</p>
      </sec>
    </sec>
    <sec id="sec5">
      <title>5. Discussion</title>
      <p>Within the designed simulation, the physician concern score was among the most discriminating linguistic features. This result is expected in part because physician concern language was generated with a probability linked to the same latent risk score used in label simulation. It therefore demonstrates that the pipeline can recover an intentionally embedded signal, not that physician concern language is independently validated as an early biomarker in real outpatient notes. Real clinicians differ substantially in documentation style, and concern language may also encode awareness of symptoms already evident at the visit [<xref ref-type="bibr" rid="B20">20</xref>][<xref ref-type="bibr" rid="B21">21</xref>].</p>
      <p>The illustrative longitudinal trajectories show patient-language features rising before simulated physician-concern language. These trajectories were generated from prespecified temporal rules and were not independently observed in patients. They should be interpreted as a hypothesis-generating demonstration of how serial-note NLP could be evaluated in a future longitudinal cohort, rather than evidence that a one- to two-week linguistic lead time exists clinically.</p>
      <p>The high specificity and modest sensitivity describe one validation-selected operating point in the synthetic test set. They do not establish an acceptable clinical alert threshold, and they should not be used to justify antidepressant interruption. In a real deployment study, thresholds would need prespecification, external calibration, prospective evaluation, and assessment of false-alert consequences.</p>
      <p>Limitations. This is a proof-of-concept simulation using fully synthetic notes and simulated labels. The same latent risk process influenced clinical meta-features, label assignment, hypomanic lexical content, and physician concern language, creating a substantial risk of simulation-induced performance advantage and circularity. Real outpatient psychiatric notes vary across clinicians, institutions, languages, templates, and documentation practices and contain missing, copied, ambiguous, and temporally misaligned information. The ClinicalBERT-sim model is not ClinicalBERT and cannot support claims about transformer contextual representations. The subgroup findings derive from simulated assignments and small test subsets. The Bayesian comparison is sensitive to the chosen effective-parameter assumptions. External validation on independently annotated real notes, preferably across institutions, is essential before clinical interpretation. Recent clinical and bipolar NLP literature also emphasises privacy, fairness, reproducibility, and domain-generalisation challenges [<xref ref-type="bibr" rid="B22">22</xref>]-[<xref ref-type="bibr" rid="B25">25</xref>].</p>
    </sec>
    <sec id="sec6">
      <title>6. Conclusion</title>
      <p>This proof-of-concept simulation study shows that an NLP pipeline combining engineered linguistic features, LSA representations, simulated clinical meta-features, and ensemble learning can recover antidepressant-associated hypomania signals intentionally embedded in synthetic outpatient-style notes. In the held-out synthetic test set, the ensemble achieved AUC = 0.912, although LightGBM achieved a slightly higher point-estimate AUC of 0.916. The analysis identifies candidate feature families and visualisation strategies for future study, but it does not validate physician concern language, hypomanic vocabulary, or any alert threshold as clinical biomarkers. The next steps are to release the complete simulator and analysis code; construct a securely governed, clinician-annotated corpus of real longitudinal BD notes; prespecify outcome timing and leakage controls; evaluate modern clinical language models against matched baselines; and perform external and prospective validation before considering EHR integration.</p>
    </sec>
    <sec id="sec7">
      <title>Author Contributions</title>
      <p>Rocco de Filippis: Conceptualization, clinical methodology, domain-specific interpretation, supervision, validation, and review. Abdullah Al Foysal: Methodology, synthetic data generation, software, natural language processing, machine-learning model development, formal analysis, visualization, and writing original draft. Both authors contributed to the interpretation of the findings, critically revised the manuscript, approved the final version, and accept accountability for the integrity of the work.</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <title>References</title>
      <ref id="B1">
        <label>1.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Graber, M.L., Franklin, N. and Gordon, R. (2005) Diagnostic Error in Internal Medicine. <italic>Archives of Internal Medicine</italic>, 165, 1493-1499. https://doi.org/10.1001/archinte.165.13.1493 <pub-id pub-id-type="doi">10.1001/archinte.165.13.1493</pub-id><pub-id pub-id-type="pmid">16009864</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1001/archinte.165.13.1493">https://doi.org/10.1001/archinte.165.13.1493</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Graber, M.L.</string-name>
              <string-name>Franklin, N.</string-name>
              <string-name>Gordon, R.</string-name>
            </person-group>
            <year>2005</year>
            <article-title>Diagnostic Error in Internal Medicine</article-title>
            <source>Archives of Internal Medicine</source>
            <volume>165</volume>
            <pub-id pub-id-type="doi">10.1001/archinte.165.13.1493</pub-id>
            <pub-id pub-id-type="pmid">16009864</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B2">
        <label>2.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Schneider, J.P. and Irons, R. (1998) Addictive Sexual Disorders: Differential Diagnosis and Treatment. <italic>Primary Psychiatry</italic>, 4, 65-70.</mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Schneider, J.P.</string-name>
              <string-name>Irons, R.</string-name>
            </person-group>
            <year>1998</year>
            <article-title>Addictive Sexual Disorders: Differential Diagnosis and Treatment</article-title>
            <source>Primary Psychiatry</source>
            <volume>4</volume>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B3">
        <label>3.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Pennebaker, J.W., Mehl, M.R. and Niederhoffer, K.G. (2003) Psychological Aspects of Natural Language Use: Our Words, Our Selves. <italic>Annual Review of Psychology</italic>, 54, 547-577. https://doi.org/10.1146/annurev.psych.54.101601.145041 <pub-id pub-id-type="doi">10.1146/annurev.psych.54.101601.145041</pub-id><pub-id pub-id-type="pmid">12185209</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1146/annurev.psych.54.101601.145041">https://doi.org/10.1146/annurev.psych.54.101601.145041</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Pennebaker, J.W.</string-name>
              <string-name>Mehl, M.R.</string-name>
              <string-name>Niederhoffer, K.G.</string-name>
              <string-name>Words, O</string-name>
            </person-group>
            <year>2003</year>
            <article-title>Psychological Aspects of Natural Language Use: Our Words, Our Selves</article-title>
            <source>Annual Review of Psychology</source>
            <volume>54</volume>
            <pub-id pub-id-type="doi">10.1146/annurev.psych.54.101601.145041</pub-id>
            <pub-id pub-id-type="pmid">12185209</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B4">
        <label>4.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Lee, J., Yoon, W., Kim, S., Kim, D., Kim, S., So, C.H., <italic>et al</italic>. (2020) BioBERT: A Pre-Trained Biomedical Language Representation Model for Biomedical Text Mining. <italic>Bioinformatics</italic>, 36, 1234-1240. https://doi.org/10.1093/bioinformatics/btz682 <pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id><pub-id pub-id-type="pmid">31501885</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bioinformatics/btz682">https://doi.org/10.1093/bioinformatics/btz682</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Lee, J.</string-name>
              <string-name>Yoon, W.</string-name>
              <string-name>Kim, S.</string-name>
              <string-name>Kim, D.</string-name>
              <string-name>Kim, S.</string-name>
              <string-name>So, C.H.</string-name>
            </person-group>
            <year>2020</year>
            <article-title>BioBERT: A Pre-Trained Biomedical Language Representation Model for Biomedical Text Mining</article-title>
            <source>Bioinformatics</source>
            <volume>36</volume>
            <pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id>
            <pub-id pub-id-type="pmid">31501885</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B5">
        <label>5.</label>
        <citation-alternatives>
          <mixed-citation publication-type="confproc">Alsentzer, E., Murphy, J., Boag, W., Weng, W., Jindi, D., Naumann, T., <italic>et al</italic>. (2019) Publicly Available Clinical. <italic>Proceedings of the</italic>2 <italic>nd Clinical Natural Language Processing Workshop</italic>, Minneapolis, 7 June 2019, 72-78. https://doi.org/10.18653/v1/w19-1909 <pub-id pub-id-type="doi">10.18653/v1/w19-1909</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.18653/v1/w19-1909">https://doi.org/10.18653/v1/w19-1909</ext-link></mixed-citation>
          <element-citation publication-type="confproc">
            <person-group person-group-type="author">
              <string-name>Alsentzer, E.</string-name>
              <string-name>Murphy, J.</string-name>
              <string-name>Boag, W.</string-name>
              <string-name>Weng, W.</string-name>
              <string-name>Jindi, D.</string-name>
              <string-name>Naumann, T.</string-name>
              <string-name>Workshop, M</string-name>
            </person-group>
            <year>2019</year>
            <article-title>Publicly Available Clinical</article-title>
            <source>Proceedings of the 2nd Clinical Natural Language Processing Workshop</source>
            <volume>7</volume>
            <pub-id pub-id-type="doi">10.18653/v1/w19-1909</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B6">
        <label>6.</label>
        <citation-alternatives>
          <mixed-citation publication-type="confproc">Vajre, V., Naylor, M., Kamath, U. and Shehu, A. (2021) PsychBERT: A Mental Health Language Model for Social Media Mental Health Behavioral Analysis. 2021 <italic>IEEE International Conference on Bioinformatics and Biomedicine</italic> ( <italic>BIBM</italic>), Houston, 9-12 December 2021, 1077-1082. https://doi.org/10.1109/bibm52615.2021.9669469 <pub-id pub-id-type="doi">10.1109/bibm52615.2021.9669469</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1109/bibm52615.2021.9669469">https://doi.org/10.1109/bibm52615.2021.9669469</ext-link></mixed-citation>
          <element-citation publication-type="confproc">
            <person-group person-group-type="author">
              <string-name>Vajre, V.</string-name>
              <string-name>Naylor, M.</string-name>
              <string-name>Kamath, U.</string-name>
              <string-name>Shehu, A.</string-name>
            </person-group>
            <year>2021</year>
            <article-title>PsychBERT: A Mental Health Language Model for Social Media Mental Health Behavioral Analysis</article-title>
            <source>2021 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</source>
            <volume>9</volume>
            <pub-id pub-id-type="doi">10.1109/bibm52615.2021.9669469</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B7">
        <label>7.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Kraepelin, E. (1921) Manic Depressive Insanity and Paranoia. <italic>The Journal of Nervous and Mental Disease</italic>, 53, 350. https://doi.org/10.1097/00005053-192104000-00057 <pub-id pub-id-type="doi">10.1097/00005053-192104000-00057</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1097/00005053-192104000-00057">https://doi.org/10.1097/00005053-192104000-00057</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Kraepelin, E.</string-name>
            </person-group>
            <year>1921</year>
            <article-title>Manic Depressive Insanity and Paranoia</article-title>
            <source>The Journal of Nervous and Mental Disease</source>
            <volume>53</volume>
            <pub-id pub-id-type="doi">10.1097/00005053-192104000-00057</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B8">
        <label>8.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Rude, S., Gortner, E. and Pennebaker, J. (2004) Language Use of Depressed and Depression-Vulnerable College Students. <italic>Cognition &amp; Emotion</italic>, 18, 1121-1133. https://doi.org/10.1080/02699930441000030 <pub-id pub-id-type="doi">10.1080/02699930441000030</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1080/02699930441000030">https://doi.org/10.1080/02699930441000030</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Rude, S.</string-name>
              <string-name>Gortner, E.</string-name>
              <string-name>Pennebaker, J.</string-name>
            </person-group>
            <year>2004</year>
            <article-title>Language Use of Depressed and Depression-Vulnerable College Students</article-title>
            <source>Cognition &amp; Emotion</source>
            <volume>18</volume>
            <pub-id pub-id-type="doi">10.1080/02699930441000030</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B9">
        <label>9.</label>
        <citation-alternatives>
          <mixed-citation publication-type="confproc">Karam, Z.N., Provost, E.M., Singh, S., Montgomery, J., Archer, C., Harrington, G., <italic>et al</italic>. (2014) Ecologically Valid Long-Term Mood Monitoring of Individuals with Bipolar Disorder Using Speech. 2014 <italic>IEEE International Conference on Acoustics</italic>, <italic>Speech and Signal Processing</italic>( <italic>ICASSP</italic>), Florence, 4-9 May 2014, 4858-4862. https://doi.org/10.1109/icassp.2014.6854525 <pub-id pub-id-type="doi">10.1109/icassp.2014.6854525</pub-id><pub-id pub-id-type="pmid">27630535</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1109/icassp.2014.6854525">https://doi.org/10.1109/icassp.2014.6854525</ext-link></mixed-citation>
          <element-citation publication-type="confproc">
            <person-group person-group-type="author">
              <string-name>Karam, Z.N.</string-name>
              <string-name>Provost, E.M.</string-name>
              <string-name>Singh, S.</string-name>
              <string-name>Montgomery, J.</string-name>
              <string-name>Archer, C.</string-name>
              <string-name>Harrington, G.</string-name>
              <string-name>Acoustics, S</string-name>
            </person-group>
            <year>2014</year>
            <article-title>Ecologically Valid Long-Term Mood Monitoring of Individuals with Bipolar Disorder Using Speech</article-title>
            <source>2014 IEEE International Conference on Acoustics</source>
            <volume>4</volume>
            <pub-id pub-id-type="doi">10.1109/icassp.2014.6854525</pub-id>
            <pub-id pub-id-type="pmid">27630535</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B10">
        <label>10.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Crema, C., Attardi, G., Sartiano, D. and Redolfi, A. (2022) Natural Language Processing in Clinical Neuroscience and Psychiatry: A Review. <italic>Frontiers in Psychiatry</italic>, 13, Article ID: 946387. https://doi.org/10.3389/fpsyt.2022.946387 <pub-id pub-id-type="doi">10.3389/fpsyt.2022.946387</pub-id><pub-id pub-id-type="pmid">36186874</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fpsyt.2022.946387">https://doi.org/10.3389/fpsyt.2022.946387</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Crema, C.</string-name>
              <string-name>Attardi, G.</string-name>
              <string-name>Sartiano, D.</string-name>
              <string-name>Redolfi, A.</string-name>
            </person-group>
            <year>2022</year>
            <article-title>Natural Language Processing in Clinical Neuroscience and Psychiatry: A Review</article-title>
            <source>Frontiers in Psychiatry</source>
            <volume>13</volume>
            <fpage>946387</fpage>
            <elocation-id>ID</elocation-id>
            <pub-id pub-id-type="doi">10.3389/fpsyt.2022.946387</pub-id>
            <pub-id pub-id-type="pmid">36186874</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B11">
        <label>11.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Harvey, D., Lobban, F., Rayson, P., Warner, A. and Jones, S. (2022) Natural Language Processing Methods and Bipolar Disorder: Scoping Review. <italic>JMIR Mental Health</italic>, 9, e35928. https://doi.org/10.2196/35928 <pub-id pub-id-type="doi">10.2196/35928</pub-id><pub-id pub-id-type="pmid">35451984</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2196/35928">https://doi.org/10.2196/35928</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Harvey, D.</string-name>
              <string-name>Lobban, F.</string-name>
              <string-name>Rayson, P.</string-name>
              <string-name>Warner, A.</string-name>
              <string-name>Jones, S.</string-name>
            </person-group>
            <year>2022</year>
            <article-title>Natural Language Processing Methods and Bipolar Disorder: Scoping Review</article-title>
            <source>JMIR Mental Health</source>
            <volume>9</volume>
            <pub-id pub-id-type="doi">10.2196/35928</pub-id>
            <pub-id pub-id-type="pmid">35451984</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B12">
        <label>12.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Hutto, A., Zikry, T.M., Bohac, B., Rose, T., Staebler, J., Slay, J., <italic>et al</italic>. (2024) Using a Natural Language Processing Toolkit to Classify Electronic Health Records by Psychiatric Diagnosis. <italic>Health Informatics Journal</italic>, 30, 1-17.</mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Hutto, A.</string-name>
              <string-name>Zikry, T.M.</string-name>
              <string-name>Bohac, B.</string-name>
              <string-name>Rose, T.</string-name>
              <string-name>Staebler, J.</string-name>
              <string-name>Slay, J.</string-name>
            </person-group>
            <year>2024</year>
            <article-title>Using a Natural Language Processing Toolkit to Classify Electronic Health Records by Psychiatric Diagnosis</article-title>
            <source>Health Informatics Journal</source>
            <volume>30</volume>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B13">
        <label>13.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Crocamo, C., Cioni, R.M., Canestro, A., Nasti, C., Palpella, D., Piacenti, S., <italic>et al</italic>. (2025) Acoustic and Natural Language Markers for Bipolar Disorder: A Pilot, mHealth Cross-Sectional Study. <italic>JMIR Formative Research</italic>, 9, e65555. https://doi.org/10.2196/65555 <pub-id pub-id-type="doi">10.2196/65555</pub-id><pub-id pub-id-type="pmid">40239203</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2196/65555">https://doi.org/10.2196/65555</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Crocamo, C.</string-name>
              <string-name>Cioni, R.M.</string-name>
              <string-name>Canestro, A.</string-name>
              <string-name>Nasti, C.</string-name>
              <string-name>Palpella, D.</string-name>
              <string-name>Piacenti, S.</string-name>
            </person-group>
            <year>2025</year>
            <article-title>Acoustic and Natural Language Markers for Bipolar Disorder: A Pilot, mHealth Cross-Sectional Study</article-title>
            <source>JMIR Formative Research</source>
            <volume>9</volume>
            <pub-id pub-id-type="doi">10.2196/65555</pub-id>
            <pub-id pub-id-type="pmid">40239203</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B14">
        <label>14.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Pennebaker, J.W., Booth, R.J. and Francis, M.E. (2007) Linguistic Inquiry and Word Count: LIWC2007. LIWC.</mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Pennebaker, J.W.</string-name>
              <string-name>Booth, R.J.</string-name>
              <string-name>Francis, M.E.</string-name>
            </person-group>
            <year>2007</year>
            <article-title>Linguistic Inquiry and Word Count: LIWC2007</article-title>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B15">
        <label>15.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Tausczik, Y.R. and Pennebaker, J.W. (2009) The Psychological Meaning of Words: LIWC and Computerized Text Analysis Methods. <italic>Journal of Language and Social Psycholog</italic><italic>y</italic>, 29, 24-54. https://doi.org/10.1177/0261927x09351676 <pub-id pub-id-type="doi">10.1177/0261927x09351676</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1177/0261927x09351676">https://doi.org/10.1177/0261927x09351676</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Tausczik, Y.R.</string-name>
              <string-name>Pennebaker, J.W.</string-name>
            </person-group>
            <year>2009</year>
            <article-title>The Psychological Meaning of Words: LIWC and Computerized Text Analysis Methods</article-title>
            <source>Journal of Language and Social Psychology</source>
            <volume>29</volume>
            <pub-id pub-id-type="doi">10.1177/0261927x09351676</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B16">
        <label>16.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Gijsman, H.J., Geddes, J.R., Rendell, J.M., Nolen, W.A. and Goodwin, G.M. (2004) Antidepressants for Bipolar Depression: A Systematic Review of Randomized, Controlled Trials. <italic>American Journal of Psychiatry</italic>, 161, 1537-1547. https://doi.org/10.1176/appi.ajp.161.9.1537 <pub-id pub-id-type="doi">10.1176/appi.ajp.161.9.1537</pub-id><pub-id pub-id-type="pmid">15337640</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1176/appi.ajp.161.9.1537">https://doi.org/10.1176/appi.ajp.161.9.1537</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Gijsman, H.J.</string-name>
              <string-name>Geddes, J.R.</string-name>
              <string-name>Rendell, J.M.</string-name>
              <string-name>Nolen, W.A.</string-name>
              <string-name>Goodwin, G.M.</string-name>
              <string-name>Randomized, C</string-name>
            </person-group>
            <year>2004</year>
            <article-title>Antidepressants for Bipolar Depression: A Systematic Review of Randomized, Controlled Trials</article-title>
            <source>American Journal of Psychiatry</source>
            <volume>161</volume>
            <pub-id pub-id-type="doi">10.1176/appi.ajp.161.9.1537</pub-id>
            <pub-id pub-id-type="pmid">15337640</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B17">
        <label>17.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Wang, Y., Wang, L., Rastegar-Mojarad, M., Moon, S., Shen, F., Afzal, N., <italic>et al</italic>. (2018) Clinical Information Extraction Applications: A Literature Review. <italic>Journal of Biomedical Informatics</italic>, 77, 34-49. https://doi.org/10.1016/j.jbi.2017.11.011 <pub-id pub-id-type="doi">10.1016/j.jbi.2017.11.011</pub-id><pub-id pub-id-type="pmid">29162496</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.jbi.2017.11.011">https://doi.org/10.1016/j.jbi.2017.11.011</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Wang, Y.</string-name>
              <string-name>Wang, L.</string-name>
              <string-name>Rastegar-Mojarad, M.</string-name>
              <string-name>Moon, S.</string-name>
              <string-name>Shen, F.</string-name>
              <string-name>Afzal, N.</string-name>
            </person-group>
            <year>2018</year>
            <article-title>Clinical Information Extraction Applications: A Literature Review</article-title>
            <source>Journal of Biomedical Informatics</source>
            <volume>77</volume>
            <pub-id pub-id-type="doi">10.1016/j.jbi.2017.11.011</pub-id>
            <pub-id pub-id-type="pmid">29162496</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B18">
        <label>18.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Deerwester, S., Dumais, S.T., Furnas, G.W., Landauer, T.K. and Harshman, R. (1990) Indexing by Latent Semantic Analysis. <italic>Journal of the American Society for Information Science</italic>, 41, 391-407. https://doi.org/10.1002/(sici)1097-4571(199009)41:6&lt;391::aid-asi1&gt;3.0.co;2-9 <pub-id pub-id-type="doi">10.1002/(sici)1097-4571(199009)41:6&lt;391::aid-asi1&gt;3.0.co;2-9</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1002/(sici)1097-4571(199009)41:6%3C391::aid-asi1%3E3.0.co;2-9">https://doi.org/10.1002/(sici)1097-4571(199009)41:6&lt;391::aid-asi1&gt;3.0.co;2-9</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Deerwester, S.</string-name>
              <string-name>Dumais, S.T.</string-name>
              <string-name>Furnas, G.W.</string-name>
              <string-name>Landauer, T.K.</string-name>
              <string-name>Harshman, R.</string-name>
            </person-group>
            <year>1990</year>
            <article-title>Indexing by Latent Semantic Analysis</article-title>
            <source>Journal of the American Society for Information Science</source>
            <volume>41</volume>
            <fpage>6</fpage>
            <pub-id pub-id-type="doi">10.1002/(sici)1097-4571(199009)41:6&lt;391::aid-asi1&gt;3.0.co;2-9</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B19">
        <label>19.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Kass, R.E. and Raftery, A.E. (1995) Bayes Factors. <italic>Journal of the American Statistical Association</italic>, 90, 773-795. https://doi.org/10.1080/01621459.1995.10476572 <pub-id pub-id-type="doi">10.1080/01621459.1995.10476572</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1080/01621459.1995.10476572">https://doi.org/10.1080/01621459.1995.10476572</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Kass, R.E.</string-name>
              <string-name>Raftery, A.E.</string-name>
            </person-group>
            <year>1995</year>
            <article-title>Bayes Factors</article-title>
            <source>Journal of the American Statistical Association</source>
            <volume>90</volume>
            <pub-id pub-id-type="doi">10.1080/01621459.1995.10476572</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B20">
        <label>20.</label>
        <citation-alternatives>
          <mixed-citation publication-type="confproc">Lundberg, S.M. and Lee, S.-I. (2017) A Unified Approach to Interpreting Model Predictions. <italic>Proceedings of the</italic>31 <italic>st International Conference on Neural Information Processing Systems</italic>, Long Beach, 4-9, December 2017, 4765-4774.</mixed-citation>
          <element-citation publication-type="confproc">
            <person-group person-group-type="author">
              <string-name>Lundberg, S.M.</string-name>
              <string-name>Lee, S.</string-name>
              <string-name>Systems, L</string-name>
            </person-group>
            <year>2017</year>
            <article-title>A Unified Approach to Interpreting Model Predictions</article-title>
            <source>Proceedings of the 31st International Conference on Neural Information Processing Systems</source>
            <volume>4</volume>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B21">
        <label>21.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Vickers, A.J. and Elkin, E.B. (2006) Decision Curve Analysis: A Novel Method for Evaluating Prediction Models. <italic>Medical Decision Making</italic>, 26, 565-574. https://doi.org/10.1177/0272989x06295361 <pub-id pub-id-type="doi">10.1177/0272989x06295361</pub-id><pub-id pub-id-type="pmid">17099194</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1177/0272989x06295361">https://doi.org/10.1177/0272989x06295361</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Vickers, A.J.</string-name>
              <string-name>Elkin, E.B.</string-name>
            </person-group>
            <year>2006</year>
            <article-title>Decision Curve Analysis: A Novel Method for Evaluating Prediction Models</article-title>
            <source>Medical Decision Making</source>
            <volume>26</volume>
            <pub-id pub-id-type="doi">10.1177/0272989x06295361</pub-id>
            <pub-id pub-id-type="pmid">17099194</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B22">
        <label>22.</label>
        <citation-alternatives>
          <mixed-citation publication-type="confproc">Chen, T. and Guestrin, C. (2016) XGBoost: A Scalable Tree Boosting System. <italic>Proceedings of the</italic>22 <italic>nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</italic>, San Francisco, 13-17 August 2016, 785-794. https://doi.org/10.1145/2939672.2939785 <pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1145/2939672.2939785">https://doi.org/10.1145/2939672.2939785</ext-link></mixed-citation>
          <element-citation publication-type="confproc">
            <person-group person-group-type="author">
              <string-name>Chen, T.</string-name>
              <string-name>Guestrin, C.</string-name>
              <string-name>Mining, S</string-name>
            </person-group>
            <year>2016</year>
            <article-title>XGBoost: A Scalable Tree Boosting System</article-title>
            <source>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>
            <volume>13</volume>
            <pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B23">
        <label>23.</label>
        <citation-alternatives>
          <mixed-citation publication-type="confproc">Ke, G., Meng, Q., Finley, T., Wang, T., Chen, W., Ma, W., Ye, Q. and Liu, T.-Y. (2017) LightGBM: A Highly Efficient Gradient Boosting Decision Tree. <italic>Proceedings of the</italic>31 <italic>st International Conference on Neural Information Processing Systems</italic>, Long Beach, 4-9 December 2017, 3149-3157.</mixed-citation>
          <element-citation publication-type="confproc">
            <person-group person-group-type="author">
              <string-name>Ke, G.</string-name>
              <string-name>Meng, Q.</string-name>
              <string-name>Finley, T.</string-name>
              <string-name>Wang, T.</string-name>
              <string-name>Chen, W.</string-name>
              <string-name>Ma, W.</string-name>
              <string-name>Ye, Q.</string-name>
              <string-name>Liu, T.</string-name>
              <string-name>Systems, L</string-name>
            </person-group>
            <year>2017</year>
            <article-title>LightGBM: A Highly Efficient Gradient Boosting Decision Tree</article-title>
            <source>Proceedings of the 31st International Conference on Neural Information Processing Systems</source>
            <volume>4</volume>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B24">
        <label>24.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">DeLong, E.R., DeLong, D.M. and Clarke-Pearson, D.L. (1988) Comparing the Areas under Two or More Correlated Receiver Operating Characteristic Curves: A Nonparametric Approach. <italic>Biometrics</italic>, 44, 837-844. https://doi.org/10.2307/2531595 <pub-id pub-id-type="doi">10.2307/2531595</pub-id><pub-id pub-id-type="pmid">3203132</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2307/2531595">https://doi.org/10.2307/2531595</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>DeLong, E.R.</string-name>
              <string-name>DeLong, D.M.</string-name>
              <string-name>Clarke-Pearson, D.L.</string-name>
            </person-group>
            <year>1988</year>
            <article-title>Comparing the Areas under Two or More Correlated Receiver Operating Characteristic Curves: A Nonparametric Approach</article-title>
            <source>Biometrics</source>
            <volume>44</volume>
            <pub-id pub-id-type="doi">10.2307/2531595</pub-id>
            <pub-id pub-id-type="pmid">3203132</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B25">
        <label>25.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Steyerberg, E.W., Vickers, A.J., Cook, N.R., Gerds, T., Gonen, M., Obuchowski, N., <italic>et al</italic>. (2010) Assessing the Performance of Prediction Models. <italic>Epidemiology</italic>, 21, 128-138. https://doi.org/10.1097/ede.0b013e3181c30fb2 <pub-id pub-id-type="doi">10.1097/ede.0b013e3181c30fb2</pub-id><pub-id pub-id-type="pmid">20010215</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1097/ede.0b013e3181c30fb2">https://doi.org/10.1097/ede.0b013e3181c30fb2</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Steyerberg, E.W.</string-name>
              <string-name>Vickers, A.J.</string-name>
              <string-name>Cook, N.R.</string-name>
              <string-name>Gerds, T.</string-name>
              <string-name>Gonen, M.</string-name>
              <string-name>Obuchowski, N.</string-name>
            </person-group>
            <year>2010</year>
            <article-title>Assessing the Performance of Prediction Models</article-title>
            <source>Epidemiology</source>
            <volume>21</volume>
            <pub-id pub-id-type="doi">10.1097/ede.0b013e3181c30fb2</pub-id>
            <pub-id pub-id-type="pmid">20010215</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
    </ref-list>
  </back>
</article>