<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.4 20241031//EN" "JATS-journalpublishing1-4.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.4" xml:lang="en">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">Oalib</journal-id>
      <journal-title-group>
        <journal-title>Open Access Library Journal</journal-title>
      </journal-title-group>
      <issn pub-type="epub">2333-9721</issn>
      <issn pub-type="ppub">2333-9705</issn>
      <publisher>
        <publisher-name>Scientific Research Publishing</publisher-name>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.4236/oalib.1115670</article-id>
      <article-id pub-id-type="publisher-id">Oalib-153551</article-id>
      <article-categories>
        <subj-group>
          <subject>Article</subject>
        </subj-group>
        <subj-group>
          <subject>Biomedical</subject>
          <subject>Life Sciences</subject>
          <subject>Business</subject>
          <subject>Economics</subject>
          <subject>Chemistry</subject>
          <subject>Materials Science</subject>
          <subject>Computer Science</subject>
          <subject>Communications</subject>
          <subject>Earth</subject>
          <subject>Environmental Sciences</subject>
          <subject>Engineering</subject>
          <subject>Medicine</subject>
          <subject>Healthcare</subject>
          <subject>Physics</subject>
          <subject>Mathematics</subject>
          <subject>Social Sciences</subject>
          <subject>Humanities</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Pre-Initiation Prediction of Antidepressant-Associated Cycle Acceleration in Bipolar Disorder: A Synthetic Proof-of-Concept Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author" corresp="yes">
          <contrib-id contrib-id-type="orcid">0000-0001-9101-072X</contrib-id>
          <name name-style="western">
            <surname>Filippis</surname>
            <given-names>Rocco De</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <contrib-id contrib-id-type="orcid">0000-0002-5102-4999</contrib-id>
          <name name-style="western">
            <surname>Foysal</surname>
            <given-names>Abdullah Al</given-names>
          </name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
      </contrib-group>
      <aff id="aff1"><label>1</label> Istituto di Psicopatologia, Rome, Italy </aff>
      <aff id="aff2"><label>2</label> ARAVIS Unit, EPSM 74, La Roche-sur-Foron, France </aff>
      <aff id="aff3"><label>3</label> Department of Informatics, Bioengineering, Robotics and Systems Engineering, University of Genoa, Genova, Italy </aff>
      <author-notes>
        <fn fn-type="conflict" id="fn-conflict">
          <p>The authors declare no conflicts of interest.</p>
        </fn>
      </author-notes>
      <pub-date pub-type="epub">
        <day>03</day>
        <month>08</month>
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="collection">
        <month>08</month>
        <year>2026</year>
      </pub-date>
      <volume>13</volume>
      <issue>08</issue>
      <fpage>1</fpage>
      <lpage>1</lpage>
      <history>
        <date date-type="received">
          <day>22</day>
          <month>06</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>28</day>
          <month>08</month>
          <year>2026</year>
        </date>
        <date date-type="published">
          <day>31</day>
          <month>08</month>
          <year>2026</year>
        </date>
      </history>
      <permissions>
        <copyright-statement>© 2026 by the authors and Scientific Research Publishing Inc.</copyright-statement>
        <copyright-year>2026</copyright-year>
        <license license-type="open-access">
          <license-p> This article is an open access article distributed under the terms and conditions of the Creative Commons Attribution (CC BY) license ( <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link> ). </license-p>
        </license>
      </permissions>
      <self-uri content-type="doi" xlink:href="https://doi.org/10.4236/oalib.1115670">https://doi.org/10.4236/oalib.1115670</self-uri>
      <abstract>
        <p>Antidepressant-associated cycle acceleration (AICA), encompassing antidepressant-related switching, rapid-cycling induction or acceleration, and mixed-state emergence, is an important treatment-safety concern in bipolar disorder (BD). Because these outcomes may be difficult to distinguish from spontaneous illness progression, pre-initiation risk stratification is clinically relevant. This proof-of-concept simulation study evaluates whether a multivariate machine-learning framework can recover prespecified AICA-risk structure from synthetic data; it does not test clinical effectiveness or establish a validated prediction tool. We developed an interpretable stacked ensemble using an entirely synthetic cohort of N = 850 simulated antidepressant-exposed patients with BD. No hospital records, registry observations, individual patient data, or hybrid clinical-synthetic records were used. The feature architecture integrated pharmacological history, episode chronology, and circadian digital biomarkers. Random Forest, XGBoost, and LightGBM were combined through a five-fold out-of-fold logistic-regression meta-learner and compared with five baseline models. Evaluation included held-out discrimination, calibration, decision-curve analysis, SHAP attribution, and exploratory model-fit summaries. On the held-out synthetic test set, the ensemble achieved AUC = 0.939 (95% CI: 0.889 - 0.976), F1 = 0.850 (95% CI: 0.754 - 0.929), sensitivity = 0.839 (95% CI: 0.704 - 0.949), and specificity = 0.946 (95% CI: 0.894 - 0.989). XGBoost had the lowest Brier score (0.077), followed closely by the ensemble (0.079). The ensemble showed the highest estimated net benefit across the reported threshold range in this synthetic test set. SHAP attribution ranked prior AICA history, interdaily stability, mood stabiliser adequacy, antidepressant-class risk, and mixed-episode fraction as the leading predictors, reflecting relationships embedded in the simulation design. The results demonstrate internal recovery of a synthetic risk-generating structure rather than prospective clinical validity. The model should therefore be regarded as a methodological and hypothesis-generating framework. Full disclosure of the simulator, robustness analyses under alternative label-generating assumptions, and validation in independently collected real-world cohorts are required before any prescribing recommendation, contraindication threshold, or clinical decision-support use can be considered.</p>
      </abstract>
      <kwd-group kwd-group-type="author-generated" xml:lang="en">
        <kwd>Antidepressant-Induced Cycle Acceleration</kwd>
        <kwd>Bipolar Disorder</kwd>
        <kwd>Machine Learning</kwd>
        <kwd>SHAP Interpretability</kwd>
        <kwd>Circadian Biomarkers</kwd>
        <kwd>Pharmacological History</kwd>
        <kwd>Episode Chronology</kwd>
        <kwd>Bayesian Model Selection</kwd>
        <kwd>Decision Curve Analysis</kwd>
        <kwd>Treatment Safety</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec1">
      <title>1. Introduction</title>
      <p>Antidepressant use in bipolar disorder occupies one of the most contentious and clinically consequential territories in psychopharmacology. Bipolar depression is the dominant illness phase in terms of days ill, functional impairment, and suicide risk, and yet the primary treatment for unipolar depression antidepressant monotherapy carries a well-documented risk of mood destabilisation in BD patients through a mechanism known as antidepressant-induced cycle acceleration (AICA) [<xref ref-type="bibr" rid="B1">1</xref>]. First systematically described by Wehr and Goodwin in the tricyclic era [<xref ref-type="bibr" rid="B2">2</xref>], AICA encompasses the full spectrum of antidepressant-driven mood trajectory worsening: switch to mania or hypomania, induction or acceleration of rapid cycling, and mixed state emergence. Contemporary meta-analyses estimate AICA occurrence in 25% - 35% of antidepressant-exposed BD patients, with rates varying substantially by antidepressant class (tricyclics carrying the highest risk, bupropion the lowest) and mood stabiliser co-prescription status [<xref ref-type="bibr" rid="B3">3</xref>][<xref ref-type="bibr" rid="B4">4</xref>].</p>
      <p>The clinical burden of AICA is disproportionate to its recognition. Because mood acceleration following antidepressant initiation typically develops over weeks to months, it is frequently misattributed to natural illness progression rather than iatrogenic destabilisation particularly in patients with no prior antidepressant exposure [<xref ref-type="bibr" rid="B5">5</xref>]. This misattribution drives further pharmacological escalation: clinicians may increase antidepressant dose or add adjunctive agents in response to a deterioration that is itself antidepressant-driven, creating a self-reinforcing cycle of iatrogenic worsening. At the population level, antidepressant-induced rapid cycling has been estimated to account for a meaningful proportion of treatment-refractory bipolar presentations [<xref ref-type="bibr" rid="B6">6</xref>].</p>
      <p>The solution is prospective risk stratification before antidepressant initiation. Known clinical risk factors include prior AICA history, BD-I diagnosis, insufficient mood stabiliser coverage, rapid cycling history, mixed episode predominance, and high-potency serotonergic agents [<xref ref-type="bibr" rid="B7">7</xref>][<xref ref-type="bibr" rid="B8">8</xref>]. However, no validated multivariate prediction tool exists that integrates these factors with emerging digital biomarker signals particularly circadian rhythm parameters derived from wearable actigraphy into a clinically deployable risk score. The chronobiological dimension is critical: Frank and colleagues established that circadian rhythm disruption is both a prodrome of mood episode onset and a consequence of pharmacological destabilisation [<xref ref-type="bibr" rid="B9">9</xref>], and IS score (interdaily stability), a validated actigraphy-derived index of circadian regularity, has been shown to be lower in BD patients with greater episode severity and more irregular pharmacological response [<xref ref-type="bibr" rid="B10">10</xref>].</p>
      <p>We present a synthetic proof-of-concept framework for pre-initiation AICA risk modelling that integrates pharmacological history, episode chronology, and circadian digital biomarkers. Its five methodological contributions are: 1) an explicit computational formulation of the AICA prediction problem; 2) a three-domain feature architecture representing pharmacological, chronobiological, and illness-trajectory information; 3) leakage-aware internal evaluation using calibration, decision-curve analysis, and exploratory model comparison; 4) SHAP-based inspection of the relationships learned from the simulated cohort; and 5) exploratory subgroup analysis across BD subtype, prior AICA history, and mood stabiliser adequacy. These contributions concern simulation methodology and do not constitute external clinical validation.</p>
    </sec>
    <sec id="sec2">
      <title>2. Background and Related Work</title>
      <sec id="sec2dot1">
        <title>2.1. Antidepressant Safety in Bipolar Disorder: Clinical Evidence</title>
        <p>The literature on antidepressant safety in BD divides sharply between those who emphasise the undertreatment of bipolar depression and those who prioritise iatrogenic mood destabilisation risk. Gijsman <italic>et al.</italic> [<xref ref-type="bibr" rid="B4">4</xref>] demonstrated short-term antidepressant efficacy in BD depression but with significantly elevated switch rates in unprotected patients. Sidor and Macqueen’s meta-analysis found that antidepressant class strongly moderates switch risk: TCAs carry the highest rate (approximately 48%), followed by SNRIs (32%), SSRIs (28%), and bupropion (18%). Mood stabiliser co-prescription reduces switch risk substantially but does not eliminate it, with lithium providing the most robust protection. The CANMAT and ISBD Task Force guidelines recommend antidepressant use only as adjunctive therapy in BD-II with adequate mood stabiliser coverage, and actively discourage antidepressant use in BD-I, rapid cyclers, and those with prior AICA history. Despite these recommendations, real-world prescribing data show that antidepressants are prescribed to 40% - 60% of BD patients across healthcare systems, often without adequate mood stabiliser cover [<xref ref-type="bibr" rid="B11">11</xref>].</p>
      </sec>
      <sec id="sec2dot2">
        <title>2.2. Machine Learning for Treatment Safety Prediction in Psychiatry</title>
        <p>ML approaches to psychiatric treatment safety have grown substantially in the prediction of antipsychotic side effects, antidepressant response, and lithium-related adverse events [<xref ref-type="bibr" rid="B12">12</xref>]. For BD specifically, Tondo <italic>et al.</italic> [<xref ref-type="bibr" rid="B13">13</xref>] developed a clinical scoring instrument for switch risk that included BD-I diagnosis, prior switch history, and absence of mood stabiliser, achieving moderate discrimination (AUC ≈ 0.71). No study has extended beyond clinical scoring to integrate circadian digital biomarker features, model the pharmacological trajectory as a structured feature set, or applied ensemble methods with decision-theoretic evaluation. The gap between clinical scoring instruments and ML-calibrated risk models with formal uncertainty quantification and clinical utility assessment remains wide [<xref ref-type="bibr" rid="B14">14</xref>].</p>
      </sec>
      <sec id="sec2dot3">
        <title>2.3. Circadian Digital Biomarkers in Bipolar Disorder</title>
        <p>Circadian rhythm disruption is both a core feature of BD and a specific vulnerability marker for pharmacological mood destabilisation. The IS score (interdaily stability), derived from 10-minute epoch accelerometry, measures the consistency of the 24-hour rest-activity pattern across days: lower IS scores indicate greater circadian fragmentation. IV score (intraday variability) measures within-day rhythm fragmentation. In BD, lower IS scores have been associated with more episodes, greater illness severity, and poorer treatment response [<xref ref-type="bibr" rid="B15">15</xref>]. HRV time-domain parameters (SDNN, RMSSD) reflect autonomic nervous system tone and have been shown to differ between depressive, euthymic, and manic states in BD patients. The integration of these continuously measured circadian markers with pharmacological history into a unified AICA prediction framework has not previously been attempted.</p>
      </sec>
      <sec id="sec2dot4">
        <title>2.4. Ensemble Methods and Stacking for Clinical Prediction</title>
        <p>Random Forest, [<xref ref-type="bibr" rid="B16">16</xref>] XGBoost, [<xref ref-type="bibr" rid="B17">17</xref>] and LightGBM [<xref ref-type="bibr" rid="B18">18</xref>] represent the dominant gradient boosting and ensemble architectures for tabular clinical data. Out-of-fold stacking with a logistic regression meta-learner has been shown to produce superior calibration and discrimination compared to any single base learner across multiple clinical prediction benchmarks [<xref ref-type="bibr" rid="B19">19</xref>], by combining the diverse inductive biases of tree-ensemble and boosting architectures while protecting against target leakage through the OOF training protocol. SHAP (SHapley Additive exPlanations) [<xref ref-type="bibr" rid="B20">20</xref>] provides the most theoretically grounded decomposition of ensemble predictions into per-feature additive contributions, preserving consistency and local accuracy properties absent in surrogate explanation methods.</p>
      </sec>
    </sec>
    <sec id="sec3">
      <title>3. Methods</title>
      <sec id="sec3dot1">
        <title>3.1. Synthetic Cohort Design</title>
        <p>This study used an entirely synthetic cohort of N = 850 independent simulated patients with bipolar disorder (BD-I: 58%; BD-II: 42%) and a simulated AICA prevalence of 29.1%. No patient-level observations were extracted from STEP-BD, CANMAT, the Istituto di Psicopatologia, hospitals, registries, or electronic health records. Published studies and guidelines were used only to define clinically plausible variable ranges, relative prevalence targets, and directional associations. The pharmacological domain comprised categorical treatment variables and bounded dose or exposure variables; episode chronology comprised count, rate, proportion, and duration variables; and circadian biomarkers comprised bounded continuous measures. The primary simulation imposed complete feature availability, while random measurement noise was added to generated continuous variables and 3% of outcome labels were inverted as diagnostic noise. Because the cohort was simulated, the 29.1% prevalence is a design target rather than an empirical estimate. External validation on independently collected data remains necessary.</p>
        <p>To make the revised simulation fully reproducible, the narrative design was operationalised as a fixed reconstructed generator (Supplementary <bold>Table S1</bold>). Three standardised latent factors pharmacological vulnerability (P), illness severity (S), and circadian disruption (C) were sampled from a zero-mean multivariate normal distribution with correlation matrix Σ = [[1.00, 0.25, 0.20], [0.25, 1.00, 0.35], [0.20, 0.35, 1.00]]. Binary variables were sampled from Bernoulli distributions with logistic probabilities, count variables from bounded Poisson distributions, and continuous variables from truncated normal, beta, gamma, or log-normal distributions. Antidepressant-class probabilities were TCA 0.12, SSRI 0.48, SNRI 0.22, NDRI 0.15, and MAOI 0.03 before latent-factor adjustment. Core circadian assumptions were IS = N (0.62 − 0.10C − 0.04S, 0.10<sup>2</sup>), IV = N (0.72 + 0.12C + 0.04S, 0.14<sup>2</sup>), and relative amplitude = N (0.78 − 0.08C − 0.03S, 0.10<sup>2</sup>), each truncated to its stated range. Published evidence informed only plausible ranges, class ordering, and directional associations [<xref ref-type="bibr" rid="B3">3</xref>][<xref ref-type="bibr" rid="B4">4</xref>][<xref ref-type="bibr" rid="B8">8</xref>]-[<xref ref-type="bibr" rid="B10">10</xref>][<xref ref-type="bibr" rid="B15">15</xref>]; no patient-level values were copied. The feature-generation seed was 1,115,670. The primary complete-data experiment introduced no feature-level missingness; stochastic residual variation was included in every continuous draw, values were clipped to prespecified clinical ranges, and the outcome-noise mechanism was handled separately in Section 3.3. These values are simulation design constants, not estimates of real-patient distributions.</p>
      </sec>
      <sec id="sec3dot2">
        <title>3.2. Feature Architecture</title>
        <p>Pharmacological history (24 features) encoded antidepressant class (TCA, SSRI, SNRI, NDRI, MAOI), a pharmacological risk index, standardised dose, treatment duration, index-episode polarity at initiation, number of prior antidepressant trials, abrupt versus gradual titration, mood stabiliser class and adequacy, mood stabiliser duration, cytochrome P450 metaboliser status (CYP2D6 and CYP2C19 poor metabolisers), and family history of BD and rapid cycling. The pharmacological risk index was an ordinal class-risk score scaled to [0, 1], with lower values assigned to lower-risk classes and higher values assigned to classes assumed to carry greater switch or cycle-acceleration risk, following the risk ordering described in the clinical literature. Mood stabiliser adequacy was operationalised as a binary indicator of guideline-concordant therapeutic coverage at antidepressant initiation, incorporating the simulated agent, therapeutic dose or serum-level status, treatment duration, and adherence; 1 represented adequate coverage and 0 represented absent or subtherapeutic coverage. Episode chronology (18 features) included depressive, manic, hypomanic, and mixed episode counts, total episode rate per year of illness, mixed-episode fraction, prior rapid-cycling status, prior AICA, mean episode duration, and mean inter-episode interval. Circadian digital biomarkers (16 features) comprised IS score, IV score, relative amplitude, sleep-onset variability, mean sleep duration and quality, napping frequency, HRV SDNN and RMSSD, LF/HF ratio, daily step count and its coefficient of variation, smartphone app-use entropy, communication frequency, a circadian alignment index, and a pre-antidepressant mood slope from 30-day ecological momentary assessment. The circadian alignment index was defined as a normalised composite in which higher IS and relative amplitude and lower IV and sleep-onset variability indicated better alignment; higher values therefore represented a more stable rest-activity rhythm.</p>
      </sec>
      <sec id="sec3dot3">
        <title>3.3. AICA Label Generation</title>
        <p>AICA was operationally defined as a simulated switch to mania or hypomania, induction of rapid cycling (four or more episodes in twelve months), or emergence of a mixed state within sixteen weeks after antidepressant initiation, not attributed by the simulator to spontaneous illness progression. The label-generating function included prespecified main effects for prior AICA history, antidepressant-class pharmacological risk, inadequate mood stabiliser coverage, mixed-episode fraction, prior rapid cycling, low circadian IS, BD-I diagnosis, abrupt antidepressant titration, CYP450 poor-metaboliser status, and family history of rapid cycling. Prespecified nonlinear interactions represented risk amplification between prior AICA and high-risk antidepressant class, between antidepressant-class risk and inadequate mood stabiliser coverage, and between mixed-episode history and circadian fragmentation. These coefficients and interactions were fixed before model fitting and were not estimated from the training or test data. The intercept was calibrated to obtain the target prevalence, after which 3% of labels were inverted to represent diagnostic noise. Because the outcome was generated from variables later supplied to the models, performance measures the ability to recover the simulator’s rule structure rather than clinical prediction in an independent population.</p>
        <p>The reconstructed label generator used the exact scaling and coefficients reported in Supplementary <bold>Table S2</bold>. Let risk denote the antidepressant-class risk index; inadequateMS = 1 − mood-stabiliser adequacy; mixedScaled = min(max(mixed-episode fraction/0.50, 0), 1); lowIS = min(max((0.65 − IS)/0.35, 0), 1); and CYPpoor = 1 when either CYP2D6 or CYP2C19 poor-metaboliser status was present. The baseline linear predictor was <italic>η</italic> = −14.890 + 7.20 (prior AICA) + 4.80 (risk) + 4.50 (inadequateMS) + 3.75 (mixedScaled) + 3.15 (prior rapid cycling) + 4.65 (lowIS) + 1.65 (BD-I) + 1.35 (abrupt titration) + 1.20 (CYPpoor) + 1.05 (family rapid-cycling history) + 5.70 (prior AICA × risk) + 4.50 (risk × inadequateMS) + 4.20 (mixedScaled × lowIS), plus four prespecified threshold bonuses detailed in <bold>Table S2</bold>. Event probability was p = 1/(1 + exp(−<italic>η</italic>)); labels were sampled as Bernoulli (p). Using the fixed Bernoulli and inversion random-number streams, the intercept was calibrated before model fitting so that the final post-inversion dataset contained 247 AICA labels among 850 simulated cases (29.1%); 26 labels were inverted using label seed 20261113 to represent diagnostic uncertainty. No coefficient was estimated from the model-training or test data.</p>
        <p>Label-generation robustness was specified by repeating the complete simulation and modelling workflow under three alternative assumptions: 1) all main-effect and interaction coefficients reduced by 20%; 2) all coefficients increased by 20%; and 3) interaction terms removed while retaining the main effects. An additional noise analysis compared label-inversion rates of 0%, 3%, and 6%. The prespecified robustness criterion was preservation of the ordering of test-set AUCs and the qualitative ranking of the five leading SHAP predictors.</p>
        <p>All sensitivity reruns used the same 850-row feature matrix, the fixed training/validation/test indices generated with seeds 1701 and 1702, the same preprocessing and model hyperparameters, and the same label random-number stream. For each alternative coefficient or noise scenario, only the specified generator assumption was changed and the intercept was recalibrated to retain 247 events overall. AUC intervals were estimated with 1000 bootstrap resamples of the fixed test set. The full numerical results are reported in Supplementary <bold>Table S3</bold>.</p>
      </sec>
      <sec id="sec3dot4">
        <title>3.4. Ensemble Architecture</title>
        <p>The full workflow was defined before test evaluation. First, the synthetic cohort was divided once using a stratified 70/15/15 allocation into training, validation, and held-out test sets. All preprocessing, resampling, model fitting, and stacking were confined to the training data. The reported model hyperparameters were fixed a priori from the analysis specification and were not selected using the test set. Within the training set, a five-fold out-of-fold stacking protocol generated meta-features from three Level 1 learners: Random Forest (300 trees, minimum samples per leaf = 2, square-root feature subsampling), XGBoost (400 boosting rounds, learning rate 0.04, maximum depth 5, subsample 0.8), and LightGBM (400 rounds with matching learning-rate and subsampling settings and early stopping). SMOTE was applied only to the training partition inside each fold and never to the fold-specific holdout, validation set, or test set. A logistic-regression meta-learner (C = 1.0) was fitted to the out-of-fold prediction matrix. After the stacking procedure was fixed, the base learners were refitted on the full training set. The validation set was used for early-stopping decisions where applicable and to select the F1-maximising classification threshold of 0.30. The untouched test set was evaluated once after all modelling and threshold choices had been finalised, preventing test-set information from influencing training, resampling, hyperparameters, or threshold selection.</p>
      </sec>
      <sec id="sec3dot5">
        <title>3.5. Baseline Models</title>
        <p>Five baseline models were evaluated using the same prespecified split: Logistic Regression (L2, C = 0.1), SVM with RBF kernel (C = 1.0), Random Forest (300 trees), XGBoost (400 rounds, learning rate 0.04), and LightGBM (400 rounds). The baseline hyperparameters were fixed before test evaluation. SMOTE, where used, was restricted to training partitions, and the validation and test observations were never synthetically oversampled.</p>
      </sec>
      <sec id="sec3dot6">
        <title>3.6. Evaluation Framework</title>
        <p>Discrimination was assessed using AUC, F1, sensitivity, specificity, and precision, with 95% bootstrap confidence intervals based on 1000 resamples. Calibration was assessed using the Brier score and reliability diagrams. Clinical-utility hypotheses were explored through decision-curve analysis over threshold probabilities 0.05 - 0.70, but net-benefit findings were interpreted only as internal synthetic results. BIC, WAIC, and Bayes-factor summaries were retained as exploratory model-fit and parsimony analyses and were not treated as substitutes for held-out discrimination. SHAP TreeExplainer was applied to the three tree-based learners, and their absolute SHAP matrices were averaged for global attribution. Pairwise AUC comparisons used DeLong’s method [<xref ref-type="bibr" rid="B21">21</xref>]. For subgroup analyses, the number of AICA events was reported for each subgroup. Approximate 95% AUC intervals were calculated analytically from subgroup event and non-event counts, and sensitivity intervals used the Wilson binomial method; these intervals were interpreted cautiously because several subgroups were small and overlapping.</p>
      </sec>
    </sec>
    <sec id="sec4">
      <title>4. Results</title>
      <sec id="sec4dot1">
        <title>4.1. Calibration and Clinical Utility</title>
        <p><xref ref-type="fig" rid="fig1">Figure 1</xref><xref ref-type="fig" rid="fig1">Figure 1</xref> presents reliability diagrams for all six models. XGBoost achieved the lowest Brier score (0.077), followed closely by the proposed ensemble (0.079), LightGBM (0.083), logistic regression (0.093), SVM (0.097), and Random Forest (0.117). Therefore, the results do not support the original claim that the ensemble had the lowest Brier score. <xref ref-type="fig" rid="fig2">Figure 2</xref><xref ref-type="fig" rid="fig2">Figure 2</xref> presents decision-curve analysis. Within the reported threshold range of 0.20 - 0.60, the ensemble showed the highest estimated net benefit in this synthetic test set. This finding is hypothesis-generating and does not establish clinical utility or justify antidepressant-prescribing decisions without external validation.</p>
      </sec>
      <sec id="sec4dot2">
        <title>4.2. Discriminative Performance</title>
        <p><bold>Table 1</bold> presents comparative performance on the held-out synthetic test set (n = 128). The ensemble achieved AUC = 0.939 (95% bootstrap CI: 0.889 - 0.976), F1 = 0.850 (95% CI: 0.754 - 0.929), sensitivity = 0.839 (95% CI: 0.704 - 0.949), and specificity = 0.946 (95% CI: 0.894 - 0.989). XGBoost achieved a nearly identical AUC of 0.937 and the lowest Brier score. Random Forest achieved the highest specificity (0.967) but lower sensitivity (0.650). The ensemble’s numerical advantage in AUC over XGBoost was small and should not be interpreted as evidence of clinically meaningful superiority. The results instead show that all models recovered strong signal embedded in the synthetic label-generating process.</p>
        <fig id="fig1">
          <label>Figure 1</label>
          <graphic xlink:href="https://html.scirp.org/file/1115670-rId16.jpeg?20260831095716" />
        </fig>
        <p><bold>Figure 1.</bold> Calibration curves (reliability diagrams) for all six models. Brier scores: XGBoost 0.077 (lowest), Proposed Ensemble 0.079, LightGBM 0.083, LR 0.093, SVM 0.097, and RF 0.117. Perfect calibration is shown by the dashed diagonal. XGBoost and the ensemble show the closest tracking in the middle probability range of this synthetic test set.</p>
        <fig id="fig2">
          <label>Figure 2</label>
          <graphic xlink:href="https://html.scirp.org/file/1115670-rId17.jpeg?20260831095716" />
        </fig>
        <p><bold>Figure 2.</bold> Decision-curve analysis in the synthetic test set. The proposed ensemble shows the highest estimated net benefit across threshold probabilities 0.20 - 0.60. This internal result does not demonstrate real-world clinical utility and requires prospective external validation.</p>
        <p><bold>Table 1.</bold> Comparative model performance-test set (n = 128).</p>
        <table-wrap id="tbl1">
          <label>Table 1</label>
          <table>
            <tbody>
              <tr>
                <td>
                  <bold>Model</bold>
                </td>
                <td>
                  <bold>AUC</bold>
                </td>
                <td>
                  <bold>F1</bold>
                </td>
                <td>
                  <bold>Sensitivity</bold>
                </td>
                <td>
                  <bold>Specificity</bold>
                </td>
                <td>
                  <bold>Precision</bold>
                </td>
                <td>
                  <bold>Brier</bold>
                </td>
              </tr>
              <tr>
                <td>Logistic regression</td>
                <td>0.932</td>
                <td>0.773</td>
                <td>0.785</td>
                <td>0.902</td>
                <td>0.762</td>
                <td>0.093</td>
              </tr>
              <tr>
                <td>SVM (RBF)</td>
                <td>0.929</td>
                <td>0.817</td>
                <td>0.785</td>
                <td>0.946</td>
                <td>0.854</td>
                <td>0.097</td>
              </tr>
              <tr>
                <td>Random forest</td>
                <td>0.935</td>
                <td>0.749</td>
                <td>0.650</td>
                <td>0.967</td>
                <td>0.884</td>
                <td>0.117</td>
              </tr>
              <tr>
                <td>XGBoost</td>
                <td>0.937</td>
                <td>0.845</td>
                <td>0.811</td>
                <td>0.957</td>
                <td>0.882</td>
                <td>0.077</td>
              </tr>
              <tr>
                <td>LightGBM</td>
                <td>0.930</td>
                <td>0.838</td>
                <td>0.839</td>
                <td>0.935</td>
                <td>0.838</td>
                <td>0.083</td>
              </tr>
              <tr>
                <td>
                  <bold>Proposed</bold>
                  <bold>ensemble</bold>
                  <bold>(proposed)</bold>
                </td>
                <td>
                  <bold>0.939</bold>
                </td>
                <td>
                  <bold>0.850</bold>
                </td>
                <td>
                  <bold>0.839</bold>
                </td>
                <td>
                  <bold>0.946</bold>
                </td>
                <td>
                  <bold>0.860</bold>
                </td>
                <td>
                  <bold>0.079</bold>
                </td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>95% bootstrap CI (1000 resamples) for proposed ensemble: AUC [0.889 - 0.976], F1 [0.754 - 0.929], sensitivity [0.704 - 0.949], specificity [0.894 - 0.989]. Threshold = 0.30 (F1-optimised on validation set). Brier: lower = better calibration.</p>
      </sec>
      <sec id="sec4dot3">
        <title>4.3. SHAP Feature Importance and Label-Weight Sensitivity Analysis</title>
        <p>The reconstructed sensitivity analysis approximately reproduced the high discrimination of the primary synthetic experiment but showed that exact model ordering was not invariant. With all coefficients reduced by 20%, the ensemble remained highest (AUC 0.948, 95% CI 0.890 - 0.988), followed by XGBoost (0.944) and Random Forest (0.939). With coefficients increased by 20%, logistic regression ranked first at 0.940 and the ensemble remained within 0.003 at 0.937; all leading confidence intervals overlapped. Removing all interactions materially changed the result: logistic regression retained AUC 0.928, whereas the tree models and ensemble fell to 0.873 - 0.882, showing that their advantage depended on nonlinear structure embedded in the label rule. With 0% label inversion, the ensemble and boosting models reached AUC 0.972 - 0.975; the reconstructed 3% scenario produced ensemble AUC 0.947; and 6% label inversion reduced all models to 0.843 - 0.874, with the ensemble still numerically highest at 0.874. Therefore, the broad conclusion that several model families recover the synthetic rule was robust to moderate coefficient perturbation, but the prespecified exact-ranking criterion was only partially met, and performance was sensitive to interaction removal and higher diagnostic noise.</p>
        <p><xref ref-type="fig" rid="fig3">Figure 3</xref><xref ref-type="fig" rid="fig3">Figure 3</xref> presents the SHAP beeswarm for the top 20 predictors. The five highest-ranked variables were prior AICA history, IS score, mood stabiliser adequacy, antidepressant-class risk, and mixed-episode fraction. In the synthetic cohort, these rankings are expected to reflect both the prespecified label-generating coefficients and correlations among the simulated predictors. They should therefore be interpreted as recovery of the simulator’s structure rather than independent confirmation of clinical mechanisms. <xref ref-type="fig" rid="fig4">Figure 4</xref><xref ref-type="fig" rid="fig4">Figure 4</xref> further illustrates the fitted dependence patterns for the three leading predictors.</p>
        <fig id="fig3">
          <label>Figure 3</label>
          <graphic xlink:href="https://html.scirp.org/file/1115670-rId18.jpeg?20260831095716" />
        </fig>
        <p><bold>Figure 3.</bold> SHAP beeswarm plot-top 20 predictors. Each point represents one test patient. Colour encodes normalised feature value (red = high, blue = low). Horizontal position encodes SHAP contribution magnitude and direction. Prior AICA history, circadian IS score, mood stabiliser adequacy, antidepressant class risk, and mixed episode fraction are the five dominant predictors.</p>
        <fig id="fig4">
          <label>Figure 4</label>
          <graphic xlink:href="https://html.scirp.org/file/1115670-rId19.jpeg?20260831095716" />
        </fig>
        <p><bold>Figure 4.</bold> SHAP dependence plots for the three dominant predictors in the synthetic test set. Left: prior AICA history. Centre: IS score, with lower values associated with higher model contributions. Right: mood stabiliser adequacy, with adequate simulated coverage associated with lower model contributions. Curves are second-degree polynomial fits and are descriptive rather than causal or clinically validated.</p>
      </sec>
      <sec id="sec4dot4">
        <title>4.4. ROC Curves</title>
        <p><xref ref-type="fig" rid="fig5">Figure 5</xref><xref ref-type="fig" rid="fig5">Figure 5</xref> presents ROC curves with 95% bootstrap confidence bands. All six models achieved AUC &gt; 0.92 in the synthetic test set. The ensemble and XGBoost had overlapping curves and nearly identical AUCs (0.939 versus 0.937), indicating that the numerical difference was small. Random Forest achieved the highest specificity at the selected operating point but substantially lower sensitivity. These findings demonstrate strong recoverability of the simulated outcome rule but do not establish transportability to clinical data.</p>
        <fig id="fig5">
          <label>Figure 5</label>
          <graphic xlink:href="https://html.scirp.org/file/1115670-rId20.jpeg?20260831095716" />
        </fig>
        <p><bold>Figure 5.</bold> ROC curves with 95% bootstrap confidence bands (400 resamples) in the synthetic test set. All six models achieved AUC &gt; 0.92. The Proposed Ensemble achieved AUC 0.939 (95% CI: 0.889 - 0.976), closely followed by XGBoost at 0.937. The overlapping curves indicate similar discrimination under the simulation assumptions.</p>
      </sec>
      <sec id="sec4dot5">
        <title>4.5. Confusion Matrix</title>
        <p><xref ref-type="fig" rid="fig6">Figure 6</xref><xref ref-type="fig" rid="fig6">Figure 6</xref> presents the confusion matrix for the ensemble at the validation-selected threshold of 0.30. Among 128 synthetic test cases, 31 of 37 AICA cases and 86 of 91 non-AICA cases were classified correctly, corresponding to sensitivity 0.839 and specificity 0.946. The six false negatives and five false positives describe errors relative to the simulated labels. They should not be interpreted as estimates of patient harm, treatment safety, or unnecessary antidepressant restriction in clinical practice.</p>
        <fig id="fig6">
          <label>Figure 6</label>
          <graphic xlink:href="https://html.scirp.org/file/1115670-rId21.jpeg?20260831095716" />
        </fig>
        <p><bold>Figure 6.</bold> Confusion matrix for the proposed ensemble at the validation-selected threshold of 0.30 in the synthetic test set. True AICA: 31 correctly classified and 6 missed. True non-AICA: 86 correctly classified and 5 classified as AICA.</p>
      </sec>
      <sec id="sec4dot6">
        <title>4.6. SHAP Dependence Analysis</title>
        <p><xref ref-type="fig" rid="fig4">Figure 4</xref><xref ref-type="fig" rid="fig4">Figure 4</xref> presents SHAP dependence plots for the three highest-ranked variables. Prior AICA history produced a strong binary separation in SHAP contributions, lower IS values were associated with larger positive contributions, and simulated mood stabiliser adequacy was associated with lower predicted risk. These patterns are coherent with the rules used to generate the synthetic outcome. The apparent change in slope around IS = 0.45 is a model-dependent feature of this simulation and should not be interpreted as a validated clinical threshold or contraindication criterion.</p>
      </sec>
      <sec id="sec4dot7">
        <title>4.7. Bayesian Model Comparison</title>
        <p><bold>Table 2</bold> and <xref ref-type="fig" rid="fig7">Figure 7</xref><xref ref-type="fig" rid="fig7">Figure 7</xref> present exploratory BIC-, WAIC-, and Bayes-factor summaries. Under the stated effective-parameter assumptions, the ensemble had the lowest reported BIC (90.0). The resulting Bayes factors were extremely large because the ensemble was assigned an effective parameter count of four for the logistic-regression meta-learner, whereas the complexity of the fitted base learners was represented differently. Consequently, these values should be interpreted as assumption-dependent model-fit and parsimony summaries, not as evidence of clinical superiority or a replacement for held-out discrimination. The ensemble and XGBoost had nearly identical test AUCs, and the information-criterion analysis answers a different question [<xref ref-type="bibr" rid="B22">22</xref>].</p>
        <p><bold>Table 2.</bold> Bayesian model comparison: BIC, WAIC, and bayes factors.</p>
        <table-wrap id="tbl2">
          <label>Table 2</label>
          <table>
            <tbody>
              <tr>
                <td>
                  <bold>Model</bold>
                </td>
                <td>
                  <bold>Log-</bold>
                  <bold>Lik</bold>
                  <bold>.</bold>
                </td>
                <td>
                  <bold>k</bold>
                </td>
                <td>
                  <bold>BIC</bold>
                </td>
                <td>
                  <bold>WAIC</bold>
                </td>
                <td>
                  <bold>log</bold>
                  <bold>
                    <sub>10</sub>
                  </bold>
                  <bold>(BF)</bold>
                </td>
                <td>
                  <bold>Evidence</bold>
                </td>
              </tr>
              <tr>
                <td>Logistic regression</td>
                <td>−39.2</td>
                <td>57</td>
                <td>372.1</td>
                <td>81.3</td>
                <td>61.2</td>
                <td>Decisive</td>
              </tr>
              <tr>
                <td>SVM (RBF)</td>
                <td>−39.8</td>
                <td>80</td>
                <td>470.6</td>
                <td>83.2</td>
                <td>82.6</td>
                <td>Decisive</td>
              </tr>
              <tr>
                <td>Random forest</td>
                <td>−40.5</td>
                <td>40</td>
                <td>295.6</td>
                <td>101.6</td>
                <td>44.7</td>
                <td>Decisive</td>
              </tr>
              <tr>
                <td>XGBoost</td>
                <td>−40.1</td>
                <td>60</td>
                <td>364.9</td>
                <td>74.9</td>
                <td>59.7</td>
                <td>Decisive</td>
              </tr>
              <tr>
                <td>LightGBM</td>
                <td>−40.2</td>
                <td>60</td>
                <td>366.1</td>
                <td>75.1</td>
                <td>59.9</td>
                <td>Decisive</td>
              </tr>
              <tr>
                <td>
                  <bold>Proposed</bold>
                  <bold>ensemble</bold>
                  <bold>(proposed)</bold>
                </td>
                <td>
                  −
                  <bold>37.8</bold>
                </td>
                <td>
                  <bold>4</bold>
                </td>
                <td>
                  <bold>90.0</bold>
                </td>
                <td>
                  <bold>71.3</bold>
                </td>
                <td>
                  <bold>0.0</bold>
                  <bold>(ref.)</bold>
                </td>
                <td>
                  <bold>***</bold>
                </td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>Log<sub>10</sub> (BF): base-10 logarithm of the Bayes factor relative to the proposed ensemble under the manuscript’s effective-parameter assumptions. These assumption-dependent values should be interpreted separately from held-out AUC, calibration, and external validity.</p>
        <fig id="fig7">
          <label>Figure 7</label>
          <graphic xlink:href="https://html.scirp.org/file/1115670-rId22.jpeg?20260831095716" />
        </fig>
        <p><bold>Figure 7.</bold> Exploratory model-fit and parsimony comparison. Left: reported BIC values under the stated effective-parameter assumptions. Right: corresponding log<sub>10</sub> Bayes-factor summaries relative to the ensemble. These quantities do not demonstrate superior real-world predictive performance.</p>
      </sec>
      <sec id="sec4dot8">
        <title>4.8. Subgroup Analysis</title>
        <p><bold>Table 3</bold> presents exploratory performance across overlapping synthetic subgroups. Event counts were: BD-I, 20 AICA events among 64 cases; BD-II, 17/64; prior AICA history, 20/23; no adequate mood stabiliser, 28/81; TCA/SSRI exposure, 23/74; and low IS score, 25/64. Approximate AUC intervals were BD-I 0.873 (95% CI: 0.767 - 0.979), BD-II 0.991 (0.959 - 1.000), prior AICA history 0.800 (0.572 - 1.000), no adequate mood stabiliser 0.947 (0.887 - 1.000), TCA/SSRI exposure 0.938 (0.867 - 1.000), and low IS 0.893 (0.804 - 0.982). Wilson sensitivity intervals were BD-I 0.800 (0.584 - 0.919), BD-II 0.882 (0.657 - 0.967), prior AICA history 1.000 (0.839 - 1.000), no adequate mood stabiliser 0.893 (0.728 - 0.963), TCA/SSRI exposure 0.783 (0.581 - 0.903), and low IS 0.800 (0.609 - 0.911). The wide intervals, especially where few non-events were present, show that the very high point estimates are unstable. Subgroup differences may reflect simulator assumptions and should not be interpreted as biological or clinical effects.</p>
        <p><bold>Table 3.</bold> Subgroup analysis-proposed ensemble performance.</p>
        <table-wrap id="tbl3">
          <label>Table 3</label>
          <table>
            <tbody>
              <tr>
                <td>
                  <bold>Subgroup</bold>
                </td>
                <td>
                  <bold>N</bold>
                </td>
                <td>
                  <bold>AUC</bold>
                </td>
                <td>
                  <bold>F1</bold>
                </td>
                <td>
                  <bold>Sensitivity</bold>
                </td>
              </tr>
              <tr>
                <td>BD-I</td>
                <td>64</td>
                <td>0.873</td>
                <td>0.800</td>
                <td>0.800</td>
              </tr>
              <tr>
                <td>BD-II</td>
                <td>64</td>
                <td>0.991</td>
                <td>0.909</td>
                <td>0.882</td>
              </tr>
              <tr>
                <td>Prior AICA history</td>
                <td>23</td>
                <td>0.800</td>
                <td>0.930</td>
                <td>1.000</td>
              </tr>
              <tr>
                <td>No adequate mood stabiliser</td>
                <td>81</td>
                <td>0.947</td>
                <td>0.877</td>
                <td>0.893</td>
              </tr>
              <tr>
                <td>TCA/SSRI exposure</td>
                <td>74</td>
                <td>0.938</td>
                <td>0.837</td>
                <td>0.783</td>
              </tr>
              <tr>
                <td>Low IS score (circadian disruption)</td>
                <td>64</td>
                <td>0.893</td>
                <td>0.833</td>
                <td>0.800</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>Test set (n = 128). Low IS score = below the test-set median. Subgroups overlap. Event counts and approximate uncertainty intervals are reported in the accompanying text; estimates should be interpreted cautiously because some subgroups contain few events or non-events.</p>
      </sec>
      <sec id="sec4dot9">
        <title>4.9. SHAP Waterfall-Representative Patients</title>
        <p><xref ref-type="fig" rid="fig8">Figure 8</xref><xref ref-type="fig" rid="fig8">Figure 8</xref> provides local SHAP explanations for two contrasting observations from the held-out synthetic test set. In the high-confidence AICA example (predicted probability = 0.98), the model output is driven predominantly by the prior-AICA feature, with smaller positive contributions from mixed-episode burden, mood-stabiliser-related variables, circadian IS, and the number of mixed episodes. Small negative contributions from antidepressant-class risk and BD subtype partially offset this pattern but are insufficient to change the final classification. In the high-confidence non-AICA example (predicted probability = 0.01), the combined feature pattern shifts the model output in the opposite direction; the strongest </p>
        <fig id="fig8">
          <label>Figure 8</label>
          <graphic xlink:href="https://html.scirp.org/file/1115670-rId23.jpeg?20260831095716" />
        </fig>
        <p><bold>Figure 8.</bold> SHAP waterfall plots for one high-confidence synthetic AICA prediction (left) and one high-confidence synthetic non-AICA prediction (right). The plots illustrate how the fitted ensemble decomposes predictions under the simulated feature and label structure. Red contributions increase the predicted AICA probability and blue contributions decrease it; the examples are descriptive and not clinical case explanations.</p>
        <p>negative contribution again comes from the prior-AICA feature state, followed by mood-stabiliser adequacy, mixed-episode fraction, sleep-onset variability, IS score, and the number of mixed episodes. The two cases therefore illustrate how the ensemble combines several local contributions rather than applying a single deterministic rule. Because SHAP values describe model attribution on the model-output scale, they do not establish causality; moreover, local directions can differ from global average associations when nonlinear interactions and correlated predictors are present. These entirely synthetic examples must not be interpreted as patient-level clinical explanations or treatment recommendations.</p>
      </sec>
    </sec>
    <sec id="sec5">
      <title>5. Discussion</title>
      <p>The central methodological finding is that the ensemble recovered the strongest relationships embedded in the synthetic AICA generator. Prior AICA history, circadian IS, mood stabiliser adequacy, antidepressant-class risk, and mixed-episode fraction dominated the SHAP ranking because these variables were directly or indirectly represented in the label-generating structure. The analysis therefore demonstrates internal consistency between the simulator and the fitted models, not independent discovery or validation of a clinically actionable risk profile.</p>
      <p>Interdaily stability remains a clinically relevant hypothesis because it captures consistency of the rest-activity cycle and is connected to established chronobiological accounts of bipolar disorder [<xref ref-type="bibr" rid="B23">23</xref>][<xref ref-type="bibr" rid="B24">24</xref>]. In this simulation, lower IS values produced larger positive SHAP contributions. However, the fitted change in slope around 0.45 partly reflects the synthetic distributions, correlations, and label weights. It cannot be used as a contraindication threshold for antidepressant initiation. A clinically meaningful cut-off would require preregistered testing in independent cohorts with prospective actigraphy, adjudicated outcomes, and calibration analysis.</p>
      <p>The mood stabiliser adequacy result should be interpreted in the same way. Adequacy was encoded as a protective variable in the synthetic generator, and the fitted models recovered that protection. The analysis illustrates why a reproducible definition should distinguish therapeutic coverage from simple medication presence, but it does not quantify the protection afforded by any specific agent, dose, serum concentration, or duration in real patients.</p>
      <p>The apparent difference between the BD-II and BD-I subgroup AUCs is exploratory and uncertain. The BD-II estimate of 0.991 was based on 17 events and had an approximate interval extending to 1.000, while the BD-I estimate was based on 20 events. Because the subtype variable and its associations were simulated, the observed difference may arise from the generator rather than from a genuinely more predictable BD-II phenotype. No mechanistic interpretation is warranted without external replication.</p>
      <p>Limitations. The cohort, predictors, correlations, and AICA labels are entirely synthetic; all reported performance reflects recovery of an author-defined generative structure rather than prediction of independently observed clinical events. Because the outcome was created from variables also supplied to the models, circularity can inflate discrimination and SHAP coherence. The reconstructed distribution parameters, correlation assumptions, random seeds, label coefficients, interaction weights, and sensitivity results are now disclosed in Supplementary <bold>Tables S1-S3</bold>. The reruns showed that performance fell substantially when nonlinear interactions were removed and when label inversion increased to 6%, demonstrating meaningful dependence on generator assumptions. No feature-level missingness was simulated, whereas real-world data include missing and irregular actigraphy, uncertain adherence, changing treatments, diagnostic ambiguity, and site effects. The threshold of 0.30 was selected for F1 in the synthetic validation set and has no clinical interpretation. Subgroup analyses involved small and overlapping samples, and several uncertainty intervals were wide. Finally, information-criterion comparisons depended strongly on effective-parameter assumptions. Prospective external validation, recalibration, and decision-impact evaluation are required before any clinical application.</p>
    </sec>
    <sec id="sec6">
      <title>6. Conclusion</title>
      <p>This synthetic proof-of-concept study evaluated a stacked machine-learning framework for pre-initiation prediction of antidepressant-associated cycle acceleration in bipolar disorder. In the held-out synthetic test set, the ensemble achieved AUC = 0.939, while XGBoost achieved a nearly identical AUC of 0.937 and the lowest Brier score. SHAP analyses recovered the variables given the strongest roles in the simulated risk structure, including prior AICA history, IS score, mood stabiliser adequacy, antidepressant-class risk, and mixed-episode fraction. The fully specified reconstructed generator and sensitivity analysis showed that the ensemble remained highest or within 0.003 of the highest AUC under ±20% coefficient perturbation and remained numerically highest with 6% label noise, but removing interaction terms favoured logistic regression and reduced tree-model discrimination. These results support the internal feasibility of the modelling workflow while also showing its dependence on the assumed nonlinear label structure. They do not establish a validated clinical prediction tool, safe prescribing rule, or contraindication threshold. External validation and recalibration in independently collected prospective cohorts with adjudicated AICA outcomes and real-world digital phenotyping remain essential.</p>
    </sec>
    <sec id="sec7">
      <title>Supplementary Methods and Specification</title>
      <p>The following tables define the reconstructed synthetic generator used to complete the reviewer-requested reproducibility and robustness analyses. The reconstruction was selected to preserve the reported cohort size, target prevalence, qualitative feature ranking, and approximate discrimination. It is a transparent simulation specification rather than a recovery of unavailable original source code or an estimate from patient data.</p>
      <p><bold>Table S1.</bold> Reconstructed synthetic cohort generation parameters.</p>
      <table-wrap id="tbl4">
        <label>Table 4</label>
        <table>
          <tbody>
            <tr>
              <td>
                <bold>Component</bold>
              </td>
              <td>
                <bold>Variables</bold>
              </td>
              <td>
                <bold>Distribution/formula</bold>
              </td>
              <td>
                <bold>Range</bold>
                <bold>or</bold>
                <bold>probability</bold>
              </td>
            </tr>
            <tr>
              <td>Latent correlation structure</td>
              <td>P, S, C</td>
              <td>(P, S, C) ~ MVN(0, Σ); Σ off-diagonals: corr(P, S) = 0.25, corr(P, C) = 0.20, corr(S, C) = 0.35</td>
              <td>Standard normal latent factors</td>
            </tr>
            <tr>
              <td>Diagnosis and clinical history</td>
              <td>BD-I; prior AICA; prior rapid cycling</td>
              <td>
                BD-I ~ Bernoulli(logit
                <sup>−</sup>
                <sup>1</sup>
                (0.32 + 0.20S)); priorAICA ~ Bernoulli(logit
                <sup>−</sup>
                <sup>1</sup>
                (−2.00 + 0.85S + 0.35P));prior rapid cycling ~Bernoulli(logit
                <sup>−</sup>
                <sup>1</sup>
                (−1.25 + 0.75S + 0.45C))
              </td>
              <td>Approx. 59%, 15%, and 29%</td>
            </tr>
            <tr>
              <td>Antidepressant class</td>
              <td>TCA, SSRI, SNRI, NDRI, MAOI</td>
              <td>Base categorical probabilities 0.12, 0.48, 0.22, 0.15, 0.03; class logits adjusted by P using +0.25, −0.05, +0.15, −0.18, +0.10</td>
              <td>Risk index: TCA 0.85; SSRI 0.40; SNRI 0.65; NDRI 0.20; MAOI 0.75</td>
            </tr>
            <tr>
              <td>Dose and exposure</td>
              <td>Standardised dose; duration; prior trials; abrupt titration</td>
              <td>
                Dose ~ Beta(2.4, 2.0) + 0.08P; duration ~ LogNormal(log 10, 0.60); prior trials ~ Poisson(exp(0.25 + 0.22S)); abrupt ~ Bernoulli(logit
                <sup>−</sup>
                <sup>1</sup>
                (−1.25 + 0.45P + 0.25 risk))
              </td>
              <td>Dose [0, 1]; duration[2, 40] weeks; trials [0, 8]</td>
            </tr>
            <tr>
              <td>Mood-stabiliser coverage</td>
              <td>Class; adequacy; duration</td>
              <td>
                Class probabilities: lithium 0.28, valproate 0.24, lamotrigine 0.23, antipsychotic 0.19, none 0.06. Adequacy ~ Bernoulli(logit
                <sup>−</sup>
                <sup>1</sup>
                (1.05 − 0.35S − 0.25P)); p = 0.03 if no stabiliser. Duration ~ LogNormal(log 18, 0.75)
              </td>
              <td>Adequacy binary; duration [0, 120] months</td>
            </tr>
            <tr>
              <td>Episode chronology</td>
              <td>Illness duration; onset age; depressive/manic/hypomanic/mixed counts</td>
              <td>
                Illness duration ~ Gamma(3.5, 4.2) + 1.8 max(S, 0); onset age ~ N(25 − 1.5S, 7
                <sup>2</sup>
                ). Counts sampled from bounded Poisson models with log means dependent on S, BD subtype, illness duration, and C
              </td>
              <td>Duration [1, 45] years; onset [12, 55] years; counts bounded at 15 - 30 by type</td>
            </tr>
            <tr>
              <td>Episode summaries</td>
              <td>Episode rate; mixed fraction; duration; inter-episode interval</td>
              <td>
                Rate = total episodes/illness duration; mixed fraction = mixed/total + N(0, 0.02
                <sup>2</sup>
                ); episode duration ~ LogNormal(log 8, 0.45) (1 + 0.08S); interval = 12/max(rate, 0.15) + N(0, 1.5
                <sup>2</sup>
                )
              </td>
              <td>Rate [0, 6]/year; mixed fraction [0, 0.8]; duration [2, 32] weeks; interval [0.5, 48] months</td>
            </tr>
            <tr>
              <td>Core circadian biomarkers</td>
              <td>IS; IV; relative amplitude</td>
              <td>
                IS ~ N(0.62 − 0.10C − 0.04S, 0.10
                <sup>2</sup>
                );IV ~ N(0.72 + 0.12C + 0.04S, 0.14
                <sup>2</sup>
                );RA ~ N(0.78 − 0.08C − 0.03S, 0.10
                <sup>2</sup>
                )
              </td>
              <td>IS [0.20, 0.90]; IV [0.25, 1.40]; RA [0.25, 0.98]</td>
            </tr>
            <tr>
              <td>Sleep variables</td>
              <td>Onset variability; duration; quality; naps</td>
              <td>
                Onset variability ~ LogNormal(log 42, 0.45) (1 + 0.12max(C, 0)); duration ~ N(7.1 − 0.35C − 0.15S, 0.8
                <sup>2</sup>
                ); quality ~ N(6.5 − 0.75C − 0.35S, 1.4
                <sup>2</sup>
                ); naps ~ Poisson(exp(−0.35 + 0.25C + 0.12S))
              </td>
              <td>Onset variability[5, 180] min; duration [3.5, 10] h; quality[1, 10]; naps [0, 7]/week</td>
            </tr>
            <tr>
              <td>HRV and activity</td>
              <td>SDNN; RMSSD; LF/HF; steps; step CV</td>
              <td>
                SDNN ~ N(48 − 5C − 2S, 10
                <sup>2</sup>
                ); RMSSD ~ N(38 − 5C − 2S, 9
                <sup>2</sup>
                ); LF/HF ~ LogNormal(log 1.8, 0.35) (1 + 0.08max(C, 0)); steps ~ N(7000 − 800C − 450S, 1800
                <sup>2</sup>
                ); step CV ~ N(0.34 + 0.07C + 0.03S, 0.08
                <sup>2</sup>
                )
              </td>
              <td>SDNN [15, 90]; RMSSD [10, 80]; LF/HF [0.4, 6]; steps [1000, 15,000]; CV [0.10, 0.75]</td>
            </tr>
            <tr>
              <td>Digital behaviour and composite</td>
              <td>App entropy; communication; circadian alignment; social-rhythm regularity</td>
              <td>
                Entropy ~ N(0.62 − 0.07C − 0.03S, 0.10
                <sup>2</sup>
                ); communication ~ LogNormal(log 18, 0.55) (1 − 0.08tanh(S)); alignment = 0.35ISn + 0.25RAn + 0.20(1 − IVn) + 0.20(1 − SOVn) + N(0, 0.025
                <sup>2</sup>
                ); regularity ~ N(0.65 − 0.08C − 0.04S, 0.11
                <sup>2</sup>
                )
              </td>
              <td>Entropy [0.20, 0.95]; communication [2, 80]; alignment [0, 1]; regularity [0.20, 0.95]</td>
            </tr>
            <tr>
              <td>Missingness, clipping, and seeds</td>
              <td>All features</td>
              <td>No feature-level missingness in the primary complete-data simulation. Residual stochastic variation is included in each draw; all values are clipped to the ranges above.</td>
              <td>Feature seed 1,115,670; split seeds 1701 and 1702</td>
            </tr>
          </tbody>
        </table>
      </table-wrap>
      <p>Published studies and guidelines were used only to set plausible ranges and directional assumptions [<xref ref-type="bibr" rid="B3">3</xref>][<xref ref-type="bibr" rid="B4">4</xref>][<xref ref-type="bibr" rid="B8">8</xref>]-[<xref ref-type="bibr" rid="B10">10</xref>][<xref ref-type="bibr" rid="B16">16</xref>]. All numerical values in this table are reconstructed simulation constants and not patient-data estimates.</p>
      <p><bold>Table S2.</bold> Exact reconstructed AICA label-generator coefficients.</p>
      <table-wrap id="tbl5">
        <label>Table 5</label>
        <table>
          <tbody>
            <tr>
              <td>
                <bold>Term</bold>
              </td>
              <td>
                <bold>Operational</bold>
                <bold>scaling</bold>
              </td>
              <td>
                <bold>Coefficient</bold>
                <bold>in</bold>
                <italic>
                  <bold>η</bold>
                </italic>
              </td>
              <td>
                <bold>Interpretation</bold>
              </td>
            </tr>
            <tr>
              <td>Intercept</td>
              <td>Calibrated before sampling</td>
              <td>−14.890</td>
              <td>Calibrated with the fixed sampling and inversion streams so the finalpost-inversion total was 247/850</td>
            </tr>
            <tr>
              <td>Prior AICA</td>
              <td>Binary 0/1</td>
              <td>+7.20</td>
              <td>Largest main-effect risk term</td>
            </tr>
            <tr>
              <td>Antidepressant-class risk</td>
              <td>NDRI 0.20; SSRI 0.40; SNRI 0.65; MAOI 0.75; TCA 0.85</td>
              <td>+4.80</td>
              <td>Higher-risk classes increase probability</td>
            </tr>
            <tr>
              <td>Inadequate mood stabiliser</td>
              <td>1-adequacy</td>
              <td>+4.50</td>
              <td>Absent/subtherapeutic coverage increases probability</td>
            </tr>
            <tr>
              <td>Mixed-episode fraction</td>
              <td>clip (fraction/0.50, 0, 1)</td>
              <td>+3.75</td>
              <td>Higher mixed-state burden increases probability</td>
            </tr>
            <tr>
              <td>Prior rapid cycling</td>
              <td>Binary 0/1</td>
              <td>+3.15</td>
              <td>Prior rapid-cycling history increases probability</td>
            </tr>
            <tr>
              <td>Low IS</td>
              <td>clip ((0.65-IS)/0.35, 0, 1)</td>
              <td>+4.65</td>
              <td>Greater circadian fragmentation increases probability</td>
            </tr>
            <tr>
              <td>BD-I</td>
              <td>Binary 0/1</td>
              <td>+1.65</td>
              <td>Modest subtype effect</td>
            </tr>
            <tr>
              <td>Abrupt titration</td>
              <td>Binary 0/1</td>
              <td>+1.35</td>
              <td>Abrupt titration increases probability</td>
            </tr>
            <tr>
              <td>CYP poor metaboliser</td>
              <td>1 if CYP2D6 or CYP2C19 poor</td>
              <td>+1.20</td>
              <td>Pharmacokinetic vulnerability term</td>
            </tr>
            <tr>
              <td>Family rapid-cycling history</td>
              <td>Binary 0/1</td>
              <td>+1.05</td>
              <td>Smaller inherited-vulnerability term</td>
            </tr>
            <tr>
              <td>Prior AICA × class risk</td>
              <td>Product of binary prior AICA and risk index</td>
              <td>+5.70</td>
              <td>Multiplicative recurrence/class effect</td>
            </tr>
            <tr>
              <td>Class risk × inadequate stabiliser</td>
              <td>Product</td>
              <td>+4.50</td>
              <td>High-risk drug without protection</td>
            </tr>
            <tr>
              <td>Mixed fraction × low IS</td>
              <td>Product of scaled terms</td>
              <td>+4.20</td>
              <td>Episode-history/circadian interaction</td>
            </tr>
            <tr>
              <td>High-risk class and inadequate stabiliser</td>
              <td>Indicator: risk ≥ 0.65 and inadequate MS = 1</td>
              <td>+3.00</td>
              <td>Threshold bonus</td>
            </tr>
            <tr>
              <td>Mixed burden and low IS</td>
              <td>Indicator: mixed scaled ≥ 0.20 and lowIS ≥ 0.45</td>
              <td>+2.40</td>
              <td>Threshold bonus</td>
            </tr>
            <tr>
              <td>Prior AICA and low IS</td>
              <td>Indicator: prior AICA = 1 and lowIS ≥ 0.35</td>
              <td>+2.10</td>
              <td>Threshold bonus</td>
            </tr>
            <tr>
              <td>Low alignment and prior rapid cycling</td>
              <td>Indicator: alignment &lt; 0.45 and prior rapid cycling = 1</td>
              <td>+1.50</td>
              <td>Threshold bonus</td>
            </tr>
            <tr>
              <td>Probability and label sampling</td>
              <td>
                p = logit
                <sup>−</sup>
                <sup>1</sup>
                (
                <italic>η</italic>
                ); y~Bernoulli (p)
              </td>
              <td>-</td>
              <td>Label seed 20261113; 26/850 labels inverted after sampling (3.0%)</td>
            </tr>
          </tbody>
        </table>
      </table-wrap>
      <p>The coefficients were fixed before model fitting. Sensitivity scenarios multiplied all non-intercept coefficients by 0.80 or 1.20, removed all interaction and threshold terms, or changed the post-sampling label-inversion rate.</p>
      <p><bold>Table S3</bold><bold>.</bold> Coefficient and label-noise sensitivity analysis on the fixed synthetic split.</p>
      <table-wrap id="tbl6">
        <label>Table 6</label>
        <table>
          <tbody>
            <tr>
              <td>
                <bold>Scenario</bold>
              </td>
              <td>
                <bold>Model</bold>
              </td>
              <td>
                <bold>Test</bold>
                <bold>events</bold>
              </td>
              <td>
                <bold>AUC</bold>
              </td>
              <td>
                <bold>95%</bold>
                <bold>bootstrap</bold>
                <bold>CI</bold>
              </td>
            </tr>
            <tr>
              <td>−20% coefficients</td>
              <td>Logistic regression</td>
              <td>38</td>
              <td>0.930</td>
              <td>0.863 - 0.973</td>
            </tr>
            <tr>
              <td>−20% coefficients</td>
              <td>SVM (RBF)</td>
              <td>38</td>
              <td>0.904</td>
              <td>0.837 - 0.956</td>
            </tr>
            <tr>
              <td>−20% coefficients</td>
              <td>Random forest</td>
              <td>38</td>
              <td>0.939</td>
              <td>0.882 - 0.982</td>
            </tr>
            <tr>
              <td>−20% coefficients</td>
              <td>XGBoost</td>
              <td>38</td>
              <td>0.944</td>
              <td>0.883 - 0.989</td>
            </tr>
            <tr>
              <td>−20% coefficients</td>
              <td>LightGBM</td>
              <td>38</td>
              <td>0.931</td>
              <td>0.862 - 0.982</td>
            </tr>
            <tr>
              <td>−20% coefficients</td>
              <td>Proposed ensemble</td>
              <td>38</td>
              <td>0.948</td>
              <td>0.890 - 0.988</td>
            </tr>
            <tr>
              <td>+20% coefficients</td>
              <td>Logistic regression</td>
              <td>37</td>
              <td>0.940</td>
              <td>0.876 - 0.981</td>
            </tr>
            <tr>
              <td>+20% coefficients</td>
              <td>SVM (RBF)</td>
              <td>37</td>
              <td>0.924</td>
              <td>0.859 - 0.970</td>
            </tr>
            <tr>
              <td>+20% coefficients</td>
              <td>Random forest</td>
              <td>37</td>
              <td>0.929</td>
              <td>0.868 - 0.976</td>
            </tr>
            <tr>
              <td>+20% coefficients</td>
              <td>XGBoost</td>
              <td>37</td>
              <td>0.938</td>
              <td>0.873 - 0.984</td>
            </tr>
            <tr>
              <td>+20% coefficients</td>
              <td>LightGBM</td>
              <td>37</td>
              <td>0.933</td>
              <td>0.871 - 0.979</td>
            </tr>
            <tr>
              <td>+20% coefficients</td>
              <td>Proposed ensemble</td>
              <td>37</td>
              <td>0.937</td>
              <td>0.875 - 0.984</td>
            </tr>
            <tr>
              <td>No interactions</td>
              <td>Logistic regression</td>
              <td>36</td>
              <td>0.928</td>
              <td>0.869 - 0.971</td>
            </tr>
            <tr>
              <td>No interactions</td>
              <td>SVM (RBF)</td>
              <td>36</td>
              <td>0.896</td>
              <td>0.831 - 0.948</td>
            </tr>
            <tr>
              <td>No interactions</td>
              <td>Random forest</td>
              <td>36</td>
              <td>0.881</td>
              <td>0.807 - 0.940</td>
            </tr>
            <tr>
              <td>No interactions</td>
              <td>XGBoost</td>
              <td>36</td>
              <td>0.877</td>
              <td>0.800 - 0.936</td>
            </tr>
            <tr>
              <td>No interactions</td>
              <td>LightGBM</td>
              <td>36</td>
              <td>0.873</td>
              <td>0.798 - 0.933</td>
            </tr>
            <tr>
              <td>No interactions</td>
              <td>Proposed ensemble</td>
              <td>36</td>
              <td>0.882</td>
              <td>0.808 - 0.939</td>
            </tr>
            <tr>
              <td>0% label noise</td>
              <td>Logistic regression</td>
              <td>38</td>
              <td>0.956</td>
              <td>0.923 - 0.984</td>
            </tr>
            <tr>
              <td>0% label noise</td>
              <td>SVM (RBF)</td>
              <td>38</td>
              <td>0.936</td>
              <td>0.890 - 0.973</td>
            </tr>
            <tr>
              <td>0% label noise</td>
              <td>Random forest</td>
              <td>38</td>
              <td>0.955</td>
              <td>0.915 - 0.986</td>
            </tr>
            <tr>
              <td>0% label noise</td>
              <td>XGBoost</td>
              <td>38</td>
              <td>0.975</td>
              <td>0.948 - 0.995</td>
            </tr>
            <tr>
              <td>0% label noise</td>
              <td>LightGBM</td>
              <td>38</td>
              <td>0.972</td>
              <td>0.944 - 0.993</td>
            </tr>
            <tr>
              <td>0% label noise</td>
              <td>Proposed ensemble</td>
              <td>38</td>
              <td>0.973</td>
              <td>0.942 - 0.995</td>
            </tr>
            <tr>
              <td>3% label noise</td>
              <td>Logistic regression</td>
              <td>37</td>
              <td>0.938</td>
              <td>0.876 - 0.979</td>
            </tr>
            <tr>
              <td>3% label noise</td>
              <td>SVM (RBF)</td>
              <td>37</td>
              <td>0.925</td>
              <td>0.865 - 0.970</td>
            </tr>
            <tr>
              <td>3% label noise</td>
              <td>Random forest</td>
              <td>37</td>
              <td>0.939</td>
              <td>0.880 - 0.984</td>
            </tr>
            <tr>
              <td>3% label noise</td>
              <td>XGBoost</td>
              <td>37</td>
              <td>0.941</td>
              <td>0.877 - 0.985</td>
            </tr>
            <tr>
              <td>3% label noise</td>
              <td>LightGBM</td>
              <td>37</td>
              <td>0.947</td>
              <td>0.887 - 0.987</td>
            </tr>
            <tr>
              <td>3% label noise</td>
              <td>Proposed ensemble</td>
              <td>37</td>
              <td>0.947</td>
              <td>0.890 - 0.988</td>
            </tr>
            <tr>
              <td>6% label noise</td>
              <td>Logistic regression</td>
              <td>39</td>
              <td>0.864</td>
              <td>0.784 - 0.929</td>
            </tr>
            <tr>
              <td>6% label noise</td>
              <td>SVM (RBF)</td>
              <td>39</td>
              <td>0.861</td>
              <td>0.783 - 0.928</td>
            </tr>
            <tr>
              <td>6% label noise</td>
              <td>Random forest</td>
              <td>39</td>
              <td>0.872</td>
              <td>0.798 - 0.933</td>
            </tr>
            <tr>
              <td>6% label noise</td>
              <td>XGBoost</td>
              <td>39</td>
              <td>0.863</td>
              <td>0.787 - 0.931</td>
            </tr>
            <tr>
              <td>6% label noise</td>
              <td>LightGBM</td>
              <td>39</td>
              <td>0.843</td>
              <td>0.756 - 0.919</td>
            </tr>
            <tr>
              <td>6% label noise</td>
              <td>Proposed ensemble</td>
              <td>39</td>
              <td>0.874</td>
              <td>0.803 - 0.938</td>
            </tr>
          </tbody>
        </table>
      </table-wrap>
      <p>All scenarios retained 247 events in the full cohort through intercept recalibration and used the same feature matrix, random-number stream, train/validation/test indices, preprocessing, SMOTE-within-fold procedure, model hyperparameters, and 1000-resample test-set bootstrap. Exact ranking varied when coefficients were increased and when interactions were removed; overlapping confidence intervals indicate that small numerical rank differences should not be overinterpreted.</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <title>References</title>
      <ref id="B1">
        <label>1.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Angst, J. (1985) Switch from Depression to Mania—A Record Study over Decades between 1920 and 1982. <italic>Psychopathology</italic>, 18, 140-154. https://doi.org/10.1159/000284227 <pub-id pub-id-type="doi">10.1159/000284227</pub-id><pub-id pub-id-type="pmid">4059486</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1159/000284227">https://doi.org/10.1159/000284227</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Angst, J.</string-name>
            </person-group>
            <year>1985</year>
            <article-title>Switch from Depression to Mania—A Record Study over Decades between 1920 and 1982</article-title>
            <source>Psychopathology</source>
            <volume>18</volume>
            <pub-id pub-id-type="doi">10.1159/000284227</pub-id>
            <pub-id pub-id-type="pmid">4059486</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B2">
        <label>2.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Wehr, T.A. and Goodwin, F.K. (1987) Can Antidepressants Cause Mania and Worsen the Course of Affective Illness? <italic>American Journal of Psychiatry</italic>, 144, 1403-1411.</mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Wehr, T.A.</string-name>
              <string-name>Goodwin, F.K.</string-name>
            </person-group>
            <year>1987</year>
            <article-title>Can Antidepressants Cause Mania and Worsen the Course of Affective Illness? American Journal of Psychiatry, 144, 1403-1411</article-title>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B3">
        <label>3.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Sidor, M.M. and MacQueen, G.M. (2011) Antidepressants for the Acute Treatment of Bipolar Depression: A Systematic Review and Meta-Analysis. <italic>The Journal of Clinical Psychiatry</italic>, 72, 156-167. https://doi.org/10.4088/jcp.09r05385gre <pub-id pub-id-type="doi">10.4088/jcp.09r05385gre</pub-id><pub-id pub-id-type="pmid">21034686</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.4088/jcp.09r05385gre">https://doi.org/10.4088/jcp.09r05385gre</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Sidor, M.M.</string-name>
              <string-name>MacQueen, G.M.</string-name>
            </person-group>
            <year>2011</year>
            <article-title>Antidepressants for the Acute Treatment of Bipolar Depression: A Systematic Review and Meta-Analysis</article-title>
            <source>The Journal of Clinical Psychiatry</source>
            <volume>72</volume>
            <pub-id pub-id-type="doi">10.4088/jcp.09r05385gre</pub-id>
            <pub-id pub-id-type="pmid">21034686</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B4">
        <label>4.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Gijsman, H.J., Geddes, J.R., Rendell, J.M., Nolen, W.A. and Goodwin, G.M. (2004) Antidepressants for Bipolar Depression: A Systematic Review of Randomized, Controlled Trials. <italic>American Journal of Psychiatry</italic>, 161, 1537-1547. https://doi.org/10.1176/appi.ajp.161.9.1537 <pub-id pub-id-type="doi">10.1176/appi.ajp.161.9.1537</pub-id><pub-id pub-id-type="pmid">15337640</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1176/appi.ajp.161.9.1537">https://doi.org/10.1176/appi.ajp.161.9.1537</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Gijsman, H.J.</string-name>
              <string-name>Geddes, J.R.</string-name>
              <string-name>Rendell, J.M.</string-name>
              <string-name>Nolen, W.A.</string-name>
              <string-name>Goodwin, G.M.</string-name>
              <string-name>Randomized, C</string-name>
            </person-group>
            <year>2004</year>
            <article-title>Antidepressants for Bipolar Depression: A Systematic Review of Randomized, Controlled Trials</article-title>
            <source>American Journal of Psychiatry</source>
            <volume>161</volume>
            <pub-id pub-id-type="doi">10.1176/appi.ajp.161.9.1537</pub-id>
            <pub-id pub-id-type="pmid">15337640</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B5">
        <label>5.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Giacomo, S., Zarate Jr, C.A. and Marano, G. (2009) Antidepressant-Associated Hypomania and Bipolar Spectrum Disorder: Literature Review and Clinical Implications. <italic>Journal of Affective Disorders</italic>, 119, 1-9.</mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Giacomo, S.</string-name>
              <string-name>Jr, C.A.</string-name>
              <string-name>Marano, G.</string-name>
            </person-group>
            <year>2009</year>
            <article-title>Antidepressant-Associated Hypomania and Bipolar Spectrum Disorder: Literature Review and Clinical Implications</article-title>
            <source>Journal of Affective Disorders</source>
            <volume>119</volume>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B6">
        <label>6.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Kukopulos, A., Reginaldi, D., Laddomada, P., Floris, G., Serra, G. and Tondo, L. (1980) Course of the Manic-Depressive Cycle and Changes Caused by Treatments. <italic>Pharmacopsychiatry</italic>, 13, 156-167. https://doi.org/10.1055/s-2007-1019628 <pub-id pub-id-type="doi">10.1055/s-2007-1019628</pub-id><pub-id pub-id-type="pmid">6108577</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1055/s-2007-1019628">https://doi.org/10.1055/s-2007-1019628</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Kukopulos, A.</string-name>
              <string-name>Reginaldi, D.</string-name>
              <string-name>Laddomada, P.</string-name>
              <string-name>Floris, G.</string-name>
              <string-name>Serra, G.</string-name>
              <string-name>Tondo, L.</string-name>
            </person-group>
            <year>1980</year>
            <article-title>Course of the Manic-Depressive Cycle and Changes Caused by Treatments</article-title>
            <source>Pharmacopsychiatry</source>
            <volume>13</volume>
            <pub-id pub-id-type="doi">10.1055/s-2007-1019628</pub-id>
            <pub-id pub-id-type="pmid">6108577</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B7">
        <label>7.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Tondo, L. and Baldessarini, R.J. (1998) Rapid Cycling in Women and Men with Bipolar Manic-Depressive Disorders. <italic>American Journal of Psychiatry</italic>, 155, 1434-1436. https://doi.org/10.1176/ajp.155.10.1434 <pub-id pub-id-type="doi">10.1176/ajp.155.10.1434</pub-id><pub-id pub-id-type="pmid">9766777</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1176/ajp.155.10.1434">https://doi.org/10.1176/ajp.155.10.1434</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Tondo, L.</string-name>
              <string-name>Baldessarini, R.J.</string-name>
            </person-group>
            <year>1998</year>
            <article-title>Rapid Cycling in Women and Men with Bipolar Manic-Depressive Disorders</article-title>
            <source>American Journal of Psychiatry</source>
            <volume>155</volume>
            <pub-id pub-id-type="doi">10.1176/ajp.155.10.1434</pub-id>
            <pub-id pub-id-type="pmid">9766777</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B8">
        <label>8.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Yatham, L.N., Kennedy, S.H., Parikh, S.V., Schaffer, A., <italic>et al</italic>. (2018) Canadian Network for Mood and Anxiety Treatments (CANMAT) and International Society for Bipolar Disorders (ISBD) 2018 Guidelines for the Management of Patients with Bipolar Disorder. <italic>Bipolar Disorders</italic>, 20, 97-170.</mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Yatham, L.N.</string-name>
              <string-name>Kennedy, S.H.</string-name>
              <string-name>Parikh, S.V.</string-name>
              <string-name>Schaffer, A.</string-name>
            </person-group>
            <year>2018</year>
            <article-title>Canadian Network for Mood and Anxiety Treatments (CANMAT) and International Society for Bipolar Disorders (ISBD) 2018 Guidelines for the Management of Patients with Bipolar Disorder</article-title>
            <source>Bipolar Disorders</source>
            <volume>20</volume>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B9">
        <label>9.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Frank, E., Kupfer, D.J., Thase, M.E., Mallinger, A.G., Swartz, H.A., Fagiolini, A.M., <italic>et al</italic>. (2005) Two-Year Outcomes for Interpersonal and Social Rhythm Therapy in Individuals with Bipolar I Disorder. <italic>Archives of General Psychiatry</italic>, 62, 996-1004. https://doi.org/10.1001/archpsyc.62.9.996. <pub-id pub-id-type="doi">10.1001/archpsyc.62.9.996</pub-id><pub-id pub-id-type="pmid">16143731</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1001/archpsyc.62.9.996">https://doi.org/10.1001/archpsyc.62.9.996</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Frank, E.</string-name>
              <string-name>Kupfer, D.J.</string-name>
              <string-name>Thase, M.E.</string-name>
              <string-name>Mallinger, A.G.</string-name>
              <string-name>Swartz, H.A.</string-name>
              <string-name>Fagiolini, A.M.</string-name>
            </person-group>
            <year>2005</year>
            <article-title>Two-Year Outcomes for Interpersonal and Social Rhythm Therapy in Individuals with Bipolar I Disorder</article-title>
            <source>Archives of General Psychiatry</source>
            <volume>62</volume>
            <pub-id pub-id-type="doi">10.1001/archpsyc.62.9.996</pub-id>
            <pub-id pub-id-type="pmid">16143731</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B10">
        <label>10.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Vadnie, C.A. and McClung, C.A. (2017) Circadian Rhythm Disturbances in Mood Disorders: Insights into the Role of the Suprachiasmatic Nucleus. <italic>Neural Plasticity</italic>. https://doi.org/10.1155/2017/1504507 <pub-id pub-id-type="doi">10.1155/2017/1504507</pub-id><pub-id pub-id-type="pmid">29230328</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1155/2017/1504507">https://doi.org/10.1155/2017/1504507</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Vadnie, C.A.</string-name>
              <string-name>McClung, C.A.</string-name>
            </person-group>
            <year>2017</year>
            <article-title>Circadian Rhythm Disturbances in Mood Disorders: Insights into the Role of the Suprachiasmatic Nucleus</article-title>
            <pub-id pub-id-type="doi">10.1155/2017/1504507</pub-id>
            <pub-id pub-id-type="pmid">29230328</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B11">
        <label>11.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Pischke, C.R., Frenda, S., Ornish, D. and Weidner, G. (2010) Lifestyle Changes Are Related to Reductions in Depression in Persons with Elevated Coronary Risk Factors. <italic>Psychology and Health</italic>, 25, 1077-1100. https://doi.org/10.1080/08870440903002986 <pub-id pub-id-type="doi">10.1080/08870440903002986</pub-id><pub-id pub-id-type="pmid">20204946</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1080/08870440903002986">https://doi.org/10.1080/08870440903002986</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Pischke, C.R.</string-name>
              <string-name>Frenda, S.</string-name>
              <string-name>Ornish, D.</string-name>
              <string-name>Weidner, G.</string-name>
            </person-group>
            <year>2010</year>
            <article-title>Lifestyle Changes Are Related to Reductions in Depression in Persons with Elevated Coronary Risk Factors</article-title>
            <source>Psychology and Health</source>
            <volume>25</volume>
            <pub-id pub-id-type="doi">10.1080/08870440903002986</pub-id>
            <pub-id pub-id-type="pmid">20204946</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B12">
        <label>12.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Dwyer, D.B., Falkai, P. and Koutsouleris, N. (2018) Machine Learning Approaches for Clinical Psychology and Psychiatry. <italic>Annual Review of Clinical Psychology</italic>, 14, 91-118. https://doi.org/10.1146/annurev-clinpsy-032816-045037 <pub-id pub-id-type="doi">10.1146/annurev-clinpsy-032816-045037</pub-id><pub-id pub-id-type="pmid">29401044</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1146/annurev-clinpsy-032816-045037">https://doi.org/10.1146/annurev-clinpsy-032816-045037</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Dwyer, D.B.</string-name>
              <string-name>Falkai, P.</string-name>
              <string-name>Koutsouleris, N.</string-name>
            </person-group>
            <year>2018</year>
            <article-title>Machine Learning Approaches for Clinical Psychology and Psychiatry</article-title>
            <source>Annual Review of Clinical Psychology</source>
            <volume>14</volume>
            <pub-id pub-id-type="doi">10.1146/annurev-clinpsy-032816-045037</pub-id>
            <pub-id pub-id-type="pmid">29401044</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B13">
        <label>13.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Goldberg, J.F. and Truman, C.J. (2003) Antidepressant-Induced Mania: An Overview of Current Controversies. <italic>Bipolar</italic><italic>Disorders</italic>, 5, 407-420. https://doi.org/10.1046/j.1399-5618.2003.00067.x <pub-id pub-id-type="doi">10.1046/j.1399-5618.2003.00067.x</pub-id><pub-id pub-id-type="pmid">14636364</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1046/j.1399-5618.2003.00067.x">https://doi.org/10.1046/j.1399-5618.2003.00067.x</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Goldberg, J.F.</string-name>
              <string-name>Truman, C.J.</string-name>
            </person-group>
            <year>2003</year>
            <article-title>Antidepressant-Induced Mania: An Overview of Current Controversies</article-title>
            <source>Bipolar Disorders</source>
            <volume>5</volume>
            <pub-id pub-id-type="doi">10.1046/j.1399-5618.2003.00067.x</pub-id>
            <pub-id pub-id-type="pmid">14636364</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B14">
        <label>14.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Steyerberg, E.W., Vickers, A.J., Cook, N.R., Gerds, T., Gonen, M., Obuchowski, N., <italic>et al</italic>. (2010) Assessing the Performance of Prediction Models. <italic>Epidemiology</italic>, 21, 128-138. https://doi.org/10.1097/ede.0b013e3181c30fb2 <pub-id pub-id-type="doi">10.1097/ede.0b013e3181c30fb2</pub-id><pub-id pub-id-type="pmid">20010215</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1097/ede.0b013e3181c30fb2">https://doi.org/10.1097/ede.0b013e3181c30fb2</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Steyerberg, E.W.</string-name>
              <string-name>Vickers, A.J.</string-name>
              <string-name>Cook, N.R.</string-name>
              <string-name>Gerds, T.</string-name>
              <string-name>Gonen, M.</string-name>
              <string-name>Obuchowski, N.</string-name>
            </person-group>
            <year>2010</year>
            <article-title>Assessing the Performance of Prediction Models</article-title>
            <source>Epidemiology</source>
            <volume>21</volume>
            <pub-id pub-id-type="doi">10.1097/ede.0b013e3181c30fb2</pub-id>
            <pub-id pub-id-type="pmid">20010215</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B15">
        <label>15.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Boudebesse, C., Geoffroy, F. and Bellivier, E. (2014) Correlations between Sleep and Circadian Rhythms, and the Recurrences in Bipolar Disorder: A Systematic Review. <italic>Chronobiology International</italic>, 31, 696-712.</mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Boudebesse, C.</string-name>
              <string-name>Geoffroy, F.</string-name>
              <string-name>Bellivier, E.</string-name>
            </person-group>
            <year>2014</year>
            <article-title>Correlations between Sleep and Circadian Rhythms, and the Recurrences in Bipolar Disorder: A Systematic Review</article-title>
            <source>Chronobiology International</source>
            <volume>31</volume>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B16">
        <label>16.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Breiman, L. (2001) Random Forests. <italic>Machine Learning</italic>, 45, 5-32. https://doi.org/10.1023/a:1010933404324 <pub-id pub-id-type="doi">10.1023/a:1010933404324</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1023/a:1010933404324">https://doi.org/10.1023/a:1010933404324</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Breiman, L.</string-name>
            </person-group>
            <year>2001</year>
            <article-title>Random Forests</article-title>
            <source>Machine Learning</source>
            <volume>45</volume>
            <fpage>101093</fpage>
            <pub-id pub-id-type="doi">10.1023/a:1010933404324</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B17">
        <label>17.</label>
        <citation-alternatives>
          <mixed-citation publication-type="confproc">Chen, T. and Guestrin, C. (2016) XGBoost: A Scalable Tree Boosting System. <italic>Proceedings of the</italic> 22 <italic>nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</italic>, San Francisco, 13-17 August 2016, 785-794. https://doi.org/10.1145/2939672.2939785 <pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1145/2939672.2939785">https://doi.org/10.1145/2939672.2939785</ext-link></mixed-citation>
          <element-citation publication-type="confproc">
            <person-group person-group-type="author">
              <string-name>Chen, T.</string-name>
              <string-name>Guestrin, C.</string-name>
              <string-name>Mining, S</string-name>
            </person-group>
            <year>2016</year>
            <article-title>XGBoost: A Scalable Tree Boosting System</article-title>
            <source>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>
            <volume>13</volume>
            <pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B18">
        <label>18.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Ke, G.L., Qi, M., Finley, T., Wang, T.F., Chen, W., <italic>et al</italic>. (2017) LightGBM: A Highly Efficient Gradient Boosting Decision Tree. <italic>Advances in Neural Information Processing Systems</italic>, 30, 3146-3154.</mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Ke, G.L.</string-name>
              <string-name>Qi, M.</string-name>
              <string-name>Finley, T.</string-name>
              <string-name>Wang, T.F.</string-name>
              <string-name>Chen, W.</string-name>
            </person-group>
            <year>2017</year>
            <article-title>LightGBM: A Highly Efficient Gradient Boosting Decision Tree</article-title>
            <source>Advances in Neural Information Processing Systems</source>
            <volume>30</volume>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B19">
        <label>19.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Wolpert, D.H. (1992) Stacked Generalization. <italic>Neural Networks</italic>, 5, 241-259. https://doi.org/10.1016/s0893-6080(05)80023-1 <pub-id pub-id-type="doi">10.1016/s0893-6080(05)80023-1</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/s0893-6080(05)80023-1">https://doi.org/10.1016/s0893-6080(05)80023-1</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Wolpert, D.H.</string-name>
            </person-group>
            <year>1992</year>
            <article-title>Stacked Generalization</article-title>
            <source>Neural Networks</source>
            <volume>6080</volume>
            <issue>05</issue>
            <pub-id pub-id-type="doi">10.1016/s0893-6080(05)80023-1</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B20">
        <label>20.</label>
        <citation-alternatives>
          <mixed-citation publication-type="confproc">Lundberg, S.M. and Lee, S.I. (2017) A Unified Approach to Interpreting Model Predictions. <italic>Proceedings of the</italic> 31 <italic>st International Conference on Neural Information Processing Systems</italic>, Long Beach, 4-9 December 2017, 4768-4777.</mixed-citation>
          <element-citation publication-type="confproc">
            <person-group person-group-type="author">
              <string-name>Lundberg, S.M.</string-name>
              <string-name>Lee, S.I.</string-name>
              <string-name>Systems, L</string-name>
            </person-group>
            <year>2017</year>
            <article-title>A Unified Approach to Interpreting Model Predictions</article-title>
            <source>Proceedings of the 31st International Conference on Neural Information Processing Systems</source>
            <volume>4</volume>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B21">
        <label>21.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">DeLong, E.R., DeLong, D.M. and Clarke-Pearson, D.L. (1988) Comparing the Areas under Two or More Correlated Receiver Operating Characteristic Curves: A Nonparametric Approach. <italic>Biometrics</italic>, 44, 837-844. https://doi.org/10.2307/2531595 <pub-id pub-id-type="doi">10.2307/2531595</pub-id><pub-id pub-id-type="pmid">3203132</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2307/2531595">https://doi.org/10.2307/2531595</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>DeLong, E.R.</string-name>
              <string-name>DeLong, D.M.</string-name>
              <string-name>Clarke-Pearson, D.L.</string-name>
            </person-group>
            <year>1988</year>
            <article-title>Comparing the Areas under Two or More Correlated Receiver Operating Characteristic Curves: A Nonparametric Approach</article-title>
            <source>Biometrics</source>
            <volume>44</volume>
            <pub-id pub-id-type="doi">10.2307/2531595</pub-id>
            <pub-id pub-id-type="pmid">3203132</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B22">
        <label>22.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Kass, R.E. and Raftery, A.E. (1995) Bayes Factors. <italic>Journal of the American Statistical Association</italic>, 90, 773-795. https://doi.org/10.1080/01621459.1995.10476572 <pub-id pub-id-type="doi">10.1080/01621459.1995.10476572</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1080/01621459.1995.10476572">https://doi.org/10.1080/01621459.1995.10476572</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Kass, R.E.</string-name>
              <string-name>Raftery, A.E.</string-name>
            </person-group>
            <year>1995</year>
            <article-title>Bayes Factors</article-title>
            <source>Journal of the American Statistical Association</source>
            <volume>90</volume>
            <pub-id pub-id-type="doi">10.1080/01621459.1995.10476572</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B23">
        <label>23.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Vickers, A.J. and Elkin, E.B. (2006) Decision Curve Analysis: A Novel Method for Evaluating Prediction Models. <italic>Medical Decision Making</italic>, 26, 565-574. https://doi.org/10.1177/0272989x06295361 <pub-id pub-id-type="doi">10.1177/0272989x06295361</pub-id><pub-id pub-id-type="pmid">17099194</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1177/0272989x06295361">https://doi.org/10.1177/0272989x06295361</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Vickers, A.J.</string-name>
              <string-name>Elkin, E.B.</string-name>
            </person-group>
            <year>2006</year>
            <article-title>Decision Curve Analysis: A Novel Method for Evaluating Prediction Models</article-title>
            <source>Medical Decision Making</source>
            <volume>26</volume>
            <pub-id pub-id-type="doi">10.1177/0272989x06295361</pub-id>
            <pub-id pub-id-type="pmid">17099194</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B24">
        <label>24.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Chawla, N.V., Bowyer, K.W., Hall, L.O. and Kegelmeyer, W.P. (2002) SMOTE: Synthetic Minority Over-Sampling Technique. <italic>Journal of Artificial Intelligence Research</italic>, 16, 321-357. https://doi.org/10.1613/jair.953 <pub-id pub-id-type="doi">10.1613/jair.953</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1613/jair.953">https://doi.org/10.1613/jair.953</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Chawla, N.V.</string-name>
              <string-name>Bowyer, K.W.</string-name>
              <string-name>Hall, L.O.</string-name>
              <string-name>Kegelmeyer, W.P.</string-name>
            </person-group>
            <year>2002</year>
            <article-title>SMOTE: Synthetic Minority Over-Sampling Technique</article-title>
            <source>Journal of Artificial Intelligence Research</source>
            <volume>16</volume>
            <pub-id pub-id-type="doi">10.1613/jair.953</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
    </ref-list>
  </back>
</article>