<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="3.0" xml:lang="en" article-type="research article">
 <front>
  <journal-meta>
   <journal-id journal-id-type="publisher-id">
    jpee
   </journal-id>
   <journal-title-group>
    <journal-title>
     Journal of Power and Energy Engineering
    </journal-title>
   </journal-title-group>
   <issn pub-type="epub">
    2327-588X
   </issn>
   <issn publication-format="print">
    2327-5901
   </issn>
   <publisher>
    <publisher-name>
     Scientific Research Publishing
    </publisher-name>
   </publisher>
  </journal-meta>
  <article-meta>
   <article-id pub-id-type="doi">
    10.4236/jpee.2025.137002
   </article-id>
   <article-id pub-id-type="publisher-id">
    jpee-144395
   </article-id>
   <article-categories>
    <subj-group subj-group-type="heading">
     <subject>
      Articles
     </subject>
    </subj-group>
    <subj-group subj-group-type="Discipline-v2">
     <subject>
      Engineering
     </subject>
    </subj-group>
   </article-categories>
   <title-group>
    An Empirical Analysis on Renewable Energy: Biogas Production Prediction Using Machine Learning
   </title-group>
   <contrib-group>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Md. Mahedi
      </surname>
      <given-names>
       Hassan
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff1"> 
      <sup>1</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Arif
      </surname>
      <given-names>
       Hossen
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff2"> 
      <sup>2</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Md. Nurunnabi
      </surname>
      <given-names>
       Sarker
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff3"> 
      <sup>3</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Yeasin
      </surname>
      <given-names>
       Arafat
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff4"> 
      <sup>4</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Aslam
      </surname>
      <given-names>
       Khan
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff1"> 
      <sup>1</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Shafiqul Islam
      </surname>
      <given-names>
       Talukder
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff5"> 
      <sup>5</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Bikash Kumar Saha
      </surname>
      <given-names>
       Roy
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff1"> 
      <sup>1</sup>
     </xref>
    </contrib>
   </contrib-group> 
   <aff id="aff1">
    <addr-line>
     aComputer Science and Engineering, World University of Bangladesh, Dhaka, Bangladesh
    </addr-line> 
   </aff> 
   <aff id="aff2">
    <addr-line>
     aBusiness Analytics, International American University, California, USA
    </addr-line> 
   </aff> 
   <aff id="aff3">
    <addr-line>
     aData Analysis, Westcliff University, California, USA
    </addr-line> 
   </aff> 
   <aff id="aff4">
    <addr-line>
     aIT Management, Westcliff University, California, USA
    </addr-line> 
   </aff> 
   <aff id="aff5">
    <addr-line>
     aComputer Science, Westcliff University, California, USA
    </addr-line> 
   </aff> 
   <pub-date pub-type="epub">
    <day>
     15
    </day> 
    <month>
     07
    </month>
    <year>
     2025
    </year>
   </pub-date> 
   <volume>
    13
   </volume> 
   <issue>
    07
   </issue>
   <fpage>
    40
   </fpage>
   <lpage>
    59
   </lpage>
   <history>
    <date date-type="received">
     <day>
      20,
     </day>
     <month>
      June
     </month>
     <year>
      2025
     </year>
    </date>
    <date date-type="published">
     <day>
      26,
     </day>
     <month>
      June
     </month>
     <year>
      2025
     </year> 
    </date> 
    <date date-type="accepted">
     <day>
      26,
     </day>
     <month>
      July
     </month>
     <year>
      2025
     </year> 
    </date>
   </history>
   <permissions>
    <copyright-statement>
     © Copyright 2014 by authors and Scientific Research Publishing Inc. 
    </copyright-statement>
    <copyright-year>
     2014
    </copyright-year>
    <license>
     <license-p>
      This work is licensed under the Creative Commons Attribution International License (CC BY). http://creativecommons.org/licenses/by/4.0/
     </license-p>
    </license>
   </permissions>
   <abstract>
    Biogas is gaining prominence as a renewable energy source with significant potential to reduce greenhouse gas emissions and mitigate environmental impacts associated with fossil fuels. This study presents an improved biogas production estimation method using machine learning (Ridge Regression, Lasso Regression, Random Forest, XGBoost, LightGBM, and GBM) combined with explainable AI (XAI) techniques to enhance model interpretability. Our rigorous evaluation using Wilcoxon Signed-Rank Tests demonstrated that LightGBM and XGBoost consistently outperformed other algorithms—LightGBM achieved superior performance in the 70:30 train-test split (RMSE = 0.075, R
    <sup>2</sup> = 0.895), while XGBoost excelled in both 80:20 (RMSE = 0.091, R
    <sup>2</sup> = 0.847) and 50:50 splits. These models proved significantly better than traditional methods (p &lt; 0.05) in all comparisons, with particularly strong performance against linear regression approaches (p-values as low as 0.001). The analysis identified waste efficiency and total daily waste (kg/day) as the most critical predictive features. Despite dataset limitations (U.S.-only livestock data, n = 344), these findings offer valuable guidance for biogas professionals and investors to optimize production forecasting and operational management strategies for improved renewable energy outputs. The research demonstrates how machine learning can enhance both prediction accuracy and interpretability in renewable energy applications.
   </abstract>
   <kwd-group> 
    <kwd>
     Biogas
    </kwd> 
    <kwd>
      Renewable Energy
    </kwd> 
    <kwd>
      Energy Production Management
    </kwd> 
    <kwd>
      Machine Learning
    </kwd> 
    <kwd>
      XAI
    </kwd> 
    <kwd>
      Analysis
    </kwd> 
    <kwd>
      Shapash
    </kwd>
   </kwd-group>
  </article-meta>
 </front>
 <body>
  <sec id="s1">
   <title>1. Introduction</title>
   <p>
    <xref ref-type="bibr" rid="scirp.144395-"></xref>Due to the refuse that domestic and industrial activities have produced, more and more well-developed and emerging nations have turned to identifying other energy sources without these sources. Not too long ago, fossil fuels were the world’s chief energy source for most of the global primary energy supply. Yet, the environmental havoc generated by fossil fuels and the reduction of natural resources have brought renewable energy sources into the spotlight as a prospective contender to ensure an environmentally sustainable future for generating energy. Interest in biogas as an alternative energy source has grown significantly over the past few years, primarily because of its ability to reduce greenhouse gas emissions. The minutes to days are the retention time, pH and compositions of medium (organic carbon and nitrogen), the temperature in the digester tank, the operating pressure of digested systems, and volatile fatty acids as essential parameters controlling biogas production yield from anaerobic digestion processes <xref ref-type="bibr" rid="scirp.144395-1">
     [1]
    </xref>. Moreover, machine learning (ML) for testing this model is a promising method to approximate complex nonlinear associations with potential outcomes. This classifies it as having the highest potential for predicting and controlling anaerobic digester performance <xref ref-type="bibr" rid="scirp.144395-2">
     [2]
    </xref>. All this is done using computer algorithms that learn from data and find insights in the absence of being explicitly programmed where to look. Numerous researchers have proposed innovative and effective strategies for modeling the biogas process using ML techniques. These methodologies encompass support vector machines, adaptive neuro-fuzzy inference systems, k-nearest neighbors, random forests, and artificial neural networks <xref ref-type="bibr" rid="scirp.144395-3">
     [3]
    </xref>. In controlled laboratory-scale experiments, three-layer artificial neural networks and nonlinear regression models have been used to predict biogas production performance <xref ref-type="bibr" rid="scirp.144395-4">
     [4]
    </xref>. Besides, adaptive neuro-fuzzy inference systems were used to simulate and optimize biogas generation from cow manure combined with maize straw in a study at pilot scale <xref ref-type="bibr" rid="scirp.144395-5">
     [5]
    </xref>. On the other hand, random forest and extreme gradient boosting (XGBoost) were successfully utilized in an industrial-scale co-digestion plant <xref ref-type="bibr" rid="scirp.144395-6">
     [6]
    </xref>. More literature should be written regarding artificial intelligence-based models for estimating biogas production and identifying essential factors influencing production from full-scale sludge digestion processes in biological treatment plants. Most researchers are focused on biogas output prediction and model building with laboratory- or pilot-scale reactors. The current study uses algorithms such as Ridge Regression, Lasso Regression, k-nearest Neighbor (KNN), ElasticNet Regression, Classification and Regression Trees, Random Forest, and finally Extreme Gradient Boosting, Light Gradient Boosting Machine, Gradient Boosting Machine, and CatBoost on U.S. biogas data. The biogas output rates were predicted using the data, and it was transferred to a fully operational anaerobic sewage digester system. The main aim of the current study is to check the efficacy of these machine learning models and validate the important variables that predict biogas production.</p>
   <p>The technical contributions of this paper are as follows:</p>
   <p>This paper part is structured in the rest of the section. As part of the related works, this reads Section 2. Section 3 presents the methodological approach we used in our experiments. This subsection will describe the methodology used to address the research problem/objective and elaborate on all methods, techniques, and tools. The results of our investigations are detailed in Section 4. The paper concludes with a summary of our findings and their implications in Section 5. In conclusion, we address the potential for future research in this discipline to be further investigated and refined.</p>
  </sec><sec id="s2">
   <title>2. Literature Review</title>
   <p>Traditional statistical methods have been implemented in numerous investigations. For example, De Clercq et al. <xref ref-type="bibr" rid="scirp.144395-7">
     [7]
    </xref> employed operations research methods, such as data envelopment analysis, and statistical techniques, such as principal component analysis and multiple linear regression, to examine the factors influencing efficient biogas projects. Their research emphasized a variety of inefficiencies, such as diminishing returns to scale. Similarly, Terradas-Ill et al. <xref ref-type="bibr" rid="scirp.144395-8">
     [8]
    </xref> created a thermal model to predict biogas production in underground, unheated fixed-dome digesters. Nevertheless, their model needed to be more suitable for large-scale facilities and needed more validation against actual data. De Clercq et al. <xref ref-type="bibr" rid="scirp.144395-9">
     [9]
    </xref> evaluated food waste and biowaste initiatives using multi-criteria decision analysis, which considered technical, economic, and environmental factors. In light of their findings, they suggested six significant policy recommendations; however, they could have offered generalized modeling tools that project administrators could employ to enhance production efficiency by utilizing waste inputs. Nevertheless, the models developed in these studies are significantly limited by their inability to incorporate the most recent developments in machine learning to predict biogas output. Instead, they depend on conventional statistical performance metrics, including root-mean-square error (RMSE) and R<sup>2</sup>. Conversely, contemporary machine learning models are assessed according to their capacity to forecast data that has not yet been observed effectively <xref ref-type="bibr" rid="scirp.144395-10">
     [10]
    </xref>. To accomplish this, datasets are partitioned into training and testing partitions, with a preference for out-of-sample evaluation metrics. These metrics are essential because they assist in the identification of potential overfitting of the model to the training data. Moreover, these conventional models are limited by a balance between simplicity and precision; as such, they cannot accurately model the complex relationships between different biochemical entities. On the other hand, ML models are inherently universal approximators of reason <xref ref-type="bibr" rid="scirp.144395-11">
     [11]
    </xref>. Since ML models have numerous tunable parameters, they can identify subtle patterns in anaerobic digestion data sets without expert guidance. Several ML-based schemes that have been used for the prediction of biogas are as follows: Wang et al. <xref ref-type="bibr" rid="scirp.144395-12">
     [12]
    </xref> built a Tree-Based Automated ML AutoML to predict biogas production from anaerobic co-digestion of degradable organic waste. Sonwai et al. <xref ref-type="bibr" rid="scirp.144395-13">
     [13]
    </xref> compared the performance of RF, XGBoost random forest, and KRR applied on three models to predict the performances of SMY. The RF model was the best, with RS = 0.85 and an RMSE = 0.06. used RF to predict a computer-generated ADM1 dataset and then compared the results with three ML systems and a random forest model. Cheon et al. <xref ref-type="bibr" rid="scirp.144395-14">
     [14]
    </xref> predicted the methane yield in an anaerobic dissolving bio-electrochemical reactor by five ML models and demonstrated the ability of the methodology to interpret complex non-linear relationships throughout several input and output significant variables in a sophisticated, complex scheme. The authors argued that an effective prediction model can aid in process stability and prevent operational hazards by using the model in real near time. De Clercq et al. <xref ref-type="bibr" rid="scirp.144395-15">
     [15]
    </xref> utilized logistic regression, SVM, and k-NN regression models on an industrial biogas facility’s data heap in China to improve the operational daily decision in ML models. Noticeably, this work did not focus on digester parameters such as temperature. A graphical user interface was used to convey the suggestions to various wastewater treatment plant engineers daily. In another study, researchers <xref ref-type="bibr" rid="scirp.144395-16">
     [16]
    </xref> compared five different ML algorithms, namely, ANN, RF, KNN, SVR, and XGBoost, on the ML to predict reliability percentile. Yildirim &amp; Ozkaya provided a benchmark of RS = 0.9242 for prediction. Most researchers have used ML because of multilayer artificial neural networks, which are highly accepted and widely commented by the engineering community, for example, optimized ANN and FCM-clustered ANFIS methodologies used to predict biogas and methane performance. FCM-ANFIS with ten clusters obtained an RS = 0.9850, MAD = 1.2463, MAPE = 5.2343, and RMSE = 1.2343, which indicates an accurate model compared to the ANN approach.</p>
   <p>A literature review shows that many researchers have used traditional statistical methods in predicting biogas production, while others have employed ANN and ML techniques <xref ref-type="bibr" rid="scirp.144395-8">
     [8]
    </xref>. However, it is necessary to utilize more suitable ML models and XAI methods, such as feature importance, to investigate the crucial levels influencing biogas yield. In this research, we intend to address these drawbacks by using more appropriate ML models and XAI methods for feature identification.</p>
  </sec><sec id="s3">
   <title>3. Methodology</title>
   <p>In this study, we propose a new top-down approach that uses machine learning techniques for preprocessing the data to predict daily biogas production using the secondary dataset. This is something new in our research because such an approach includes numerous explainable artificial intelligence (XAI) models to find the most substantial factors impacting biogas production.</p>
   <sec id="s3_1">
    <title>3.1. Overview of the Methodology</title>
    <p>This study develops a machine learning methodology for biogas production prediction using operational data from U.S. livestock facilities, implementing a rigorous analytical pipeline that begins with comprehensive data preprocessing, including ordinal and one-hot encoding, alongside data normalization demonstrated in <xref ref-type="fig" rid="fig1">
      Figure 1
     </xref>. We systematically evaluate model performance across three train-test splits (80:20, 70:30, and 50:50) incorporating both traditional regression techniques (Ridge, Lasso) and advanced ensemble methods (Random Forest, GBM, XGBoost, LightGBM), with hyperparameter optimization and cross-validation to ensure robustness. The evaluation employs RMSE and R<sup>2</sup> metrics complemented by Wilcoxon signed-rank tests to statistically verify performance differences, while SHAP analysis provides interpretability by identifying key predictive features like waste inputs and their directional impacts. This integrated approach combines predictive modeling with explainable AI techniques and offers both accurate forecasts and operational insights while addressing potential overfitting through regularization and extensive validation across multiple data partitions despite the limitations of a U.S.-specific dataset with a moderate sample size.</p>
    <fig id="fig1" position="float">
     <label>Figure 1</label>
     <caption>
      <title>Figure 1. Overview of the proposed methodology.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/1771231-rId12.jpeg?20250729024204" />
    </fig>
   </sec>
   <sec id="s3_2">
    <title>3.2. Description of Dataset and Variables</title>
    <p>This study uses data from the U.S. Biogas dataset on Kaggle to forecast biogas production at regular intervals <xref ref-type="bibr" rid="scirp.144395-17">
      [17]
     </xref>. Covering cattle farms from around the U.S., this large dataset paints a broad picture of how biogas is produced. A vital tool for understanding when and how much renewable energy may be usable. It has data from cattle, dairy cows, and pig biogas projects with chickens. It can prove handy for people from different professions, including agriculture, green energy, environmental policy, etc., displayed in <xref ref-type="table" rid="table1">
      Table 1
     </xref>. The dataset consists of 491 observations from various places in the United States and about 29 attributes. Data preparation is one of the vital parts of machine learning, but this step consumes too much time work (around 60% of the budget for a data science project) <xref ref-type="bibr" rid="scirp.144395-2">
      [2]
     </xref>. Missing values were handled through median imputation for numerical data and mode imputation for categorical data. Outliers were removed using the IQR method. Categorical variables were encoded using one-hot encoding.</p>
    <table-wrap id="table1">
     <label>
      <xref ref-type="table" rid="table1">
       Table 1
      </xref></label>
     <caption>
      <title>
       <xref ref-type="bibr" rid="scirp.144395-"></xref>Table 1. A brief description of the dataset.</title>
     </caption>
     <table class="MsoTableGrid custom-table" border="0" cellspacing="0" cellpadding="0"> 
      <tr> 
       <td class="custom-bottom-td custom-top-td acenter" width="51.55%"><p style="text-align:center">Column name</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="55.62%"><p style="text-align:center">Description</p></td> 
      </tr> 
      <tr> 
       <td class="custom-top-td acenter" width="51.55%"><p style="text-align:center">Year Operational</p></td> 
       <td class="custom-top-td acenter" width="55.62%"><p style="text-align:center">The year when the project became operational.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Cattle</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Number of cattle involved.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Dairy</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Number of dairy cows involved.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Poultry</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Number of poultry involved.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Swine</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Number of swine involved.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Biogas Generation Estimate (cu-ft/day)</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Estimated daily biogas production.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Electricity Generated (kWh/yr)</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Estimated annual electricity generation.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Total Emission Reductions (MTCO2e/yr)</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Estimated total emission reduction.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Operational Years</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Number of years the project has been operational.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Total_Animals</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Total number of animals involved in the project.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Biogas_per_Animal (cu-ft/day)</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Estimated biogas production per animal.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Emission_Reduction_per_Year</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Estimated annual emission reduction per animal.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Electricity_to_Biogas_Ratio</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">The ratio between electricity generation and biogas production.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Total_Waste_kg/day</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Estimated daily waste production.</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="51.55%"><p style="text-align:center">Waste_Efficiency</p></td> 
       <td class="acenter" width="55.62%"><p style="text-align:center">Efficiency of waste conversion to biogas.</p></td> 
      </tr> 
      <tr> 
       <td class="custom-bottom-td acenter" width="51.55%"><p style="text-align:center">Electricity_Efficiency</p></td> 
       <td class="custom-bottom-td acenter" width="55.62%"><p style="text-align:center">Efficiency of biogas conversion to electricity.</p></td> 
      </tr> 
     </table>
    </table-wrap>
   </sec>
   <sec id="s3_3">
    <title>3.3. Machine Learning Algorithms</title>
    <p>Lasso regression, or Least Absolute Shrinkage and Selection Operator, is a regularization technique used to enhance the prediction accuracy and interpretability of regression models by enforcing sparsity. Unlike ridge regression, which applies an L<sub>2</sub> penalty, lasso regression adds an L<sub>1</sub> penalty to the loss function, where Y<sub>i</sub> are the observed values, 
     <math xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <mover accent="true"> 
        <mrow> 
         <mi>
           Y 
         </mi> 
         <mi>
           i 
         </mi> 
        </mrow> 
        <mo stretchy="true">
          ^ 
        </mo> 
       </mover> 
      </mrow> 
     </math> that predicted values, βj the coefficients, and λ the regularization parameter <xref ref-type="bibr" rid="scirp.144395-18">
      [18]
     </xref>. The L<sub>1 </sub>penalty tends to shrink some coefficients exactly to zero, effectively performing variable selection and yielding a simpler model that retains only the most significant predictors. This sparsity property makes lasso regression particularly useful when dealing with high-dimensional data where the number of predictors exceeds the number of observations <xref ref-type="bibr" rid="scirp.144395-19">
      [19]
     </xref>. By appropriately tuning the λ parameter, one can control the complexity of the model, balancing bias and variance to improve predictive performance and interpretability <xref ref-type="bibr" rid="scirp.144395-20">
      [20]
     </xref>. Lasso regression is widely utilized in fields like bioinformatics and economics where model simplicity and feature selection are crucial <xref ref-type="bibr" rid="scirp.144395-21">
      [21]
     </xref>.</p>
    <p>Ridge regression is an extension of linear regression that addresses multicollinearity among predictor variables by incorporating a regularization term. This technique adds a penalty equal to the square of the magnitude of the coefficients to the loss function, thus shrinking the coefficients and reducing their variance <xref ref-type="bibr" rid="scirp.144395-22">
      [22]
     </xref>. The ridge regression equation modifies the ordinary least squares (OLS) regression by adding a regularization parameter λ, which minimizes the following cost function where Y<sub>i</sub> represents the observed values, Y<sub>i</sub> the predicted values, βj the coefficients, and λ the regularization parameter. By tuning λ, one can control the trade-off between fitting the data well and keeping the model coefficients small, which helps mitigate overfitting. Ridge regression is instrumental in situations with many correlated predictors, as it improves the model’s generalization performance <xref ref-type="bibr" rid="scirp.144395-20">
      [20]
     </xref>. This method retains all predictors in the final model, unlike other techniques such as Lasso regression, which can shrink some coefficients to zero <xref ref-type="bibr" rid="scirp.144395-19">
      [19]
     </xref>.</p>
    <p>Gradient Boosting Machine (GBM) is an ensemble learning technique that builds a strong predictive model by sequentially combining multiple weak learners, typically decision trees, to minimize a predefined loss function. GBM constructs each tree iteratively, focusing on reducing the errors made by the previous trees. At each iteration, GBM fits a new tree to the residuals or negative gradients of the loss function from the ensemble of previously built trees. The new tree is then added to the ensemble, with its predictions weighted based on a learning rate parameter. This process continues iteratively until a predefined number of trees is reached or until further iterations cease to improve performance <xref ref-type="bibr" rid="scirp.144395-23">
      [23]
     </xref>. GBM is highly flexible and capable of capturing complex nonlinear relationships in the data. However, it is sensitive to hyperparameters such as the learning rate, tree depth, and the number of trees in the ensemble. Careful tuning of these hyperparameters is essential to prevent overfitting and achieve optimal performance. GBM has achieved remarkable success in various ML competitions and is widely used in practice for tasks such as classification, regression, and ranking <xref ref-type="bibr" rid="scirp.144395-24">
      [24]
     </xref>.</p>
    <p>Random Forest is an ensemble learning method that combines multiple decision trees to improve prediction accuracy and generalization performance. Each decision tree in the forest is trained on a bootstrap sample of the training data and selects the best feature from a random subset of features at each split. The final prediction in Random Forest is typically made by averaging or taking a majority vote of the predictions from individual trees <xref ref-type="bibr" rid="scirp.144395-25">
      [25]
     </xref>. Random Forest offers several advantages over a single decision tree. Firstly, it reduces overfitting by averaging predictions from multiple trees, thereby improving the model’s ability to generalize to unseen data. Secondly, it provides estimates of feature importance, which can help identify the most informative features in the dataset. The key hyperparameters in Random Forest include the number of trees in the forest Ntrees and the size of the random feature subset considered at each split m. These hyperparameters influence the model’s bias-variance trade-off: increasing Ntrees typically reduces variance but may increase computational cost, while increasing m can decrease correlation between trees but may increase bias. Random Forest is robust to noisy data and can handle high-dimensional feature spaces, making it a popular choice for classification and regression tasks in various domains <xref ref-type="bibr" rid="scirp.144395-19">
      [19]
     </xref>. However, its interpretability is lower compared to single decision trees, as it is more challenging to interpret the combined predictions of multiple trees <xref ref-type="bibr" rid="scirp.144395-26">
      [26]
     </xref>.</p>
    <p>LightGBM is a gradient boosting framework developed by Microsoft that focuses on efficiency, speed, and accuracy. It is designed to handle large-scale datasets and can be significantly faster than other gradient boosting implementations, making it well-suited for both industry-scale applications and research <xref ref-type="bibr" rid="scirp.144395-27">
      [27]
     </xref>. LightGBM adopts a novel approach to tree construction known as Gradient-based One-Side Sampling (GOSS) and Exclusive Feature Bundling (EFB), which enable it to achieve faster training times and lower memory usage without sacrificing predictive performance. Additionally, LightGBM supports parallel and distributed computing, allowing it to scale efficiently to multi-core CPUs and distributed computing environments <xref ref-type="bibr" rid="scirp.144395-28">
      [28]
     </xref>. The key hyperparameters in LightGBM include the learning rate η, tree depth maxdepth, number of leaves num_leaves, and regularization parameters λ and ɑ. These hyperparameters influence the model’s capacity, complexity, and generalization ability and must be carefully tuned to optimize performance. LightGBM offers native support for categorical features and missing values, which simplifies data preprocessing and feature engineering tasks. It supports both classification and regression tasks and can handle various loss functions, including binary classification, multi-class classification, and regression. LightGBM has become a popular choice among ML practitioners and researchers due to its excellent performance, scalability, and ease of use.</p>
    <p>
     <xref ref-type="bibr" rid="scirp.144395-"></xref>XGBoost, short for eXtreme Gradient Boosting, is an optimized distributed gradient boosting library designed for efficiency, scalability, and flexibility <xref ref-type="bibr" rid="scirp.144395-29">
      [29]
     </xref>. It is an extension of the gradient boosting framework that focuses on computational speed and model performance. XGBoost has gained widespread popularity in ML competitions and real-world applications due to its state-of-the-art performance and robustness. XGBoost employs a boosted ensemble of decision trees, where each tree is trained sequentially to correct the errors made by the previous ones. It incorporates several innovative techniques, such as parallelized tree construction, approximate tree learning, and regularization, to improve training speed and accuracy <xref ref-type="bibr" rid="scirp.144395-24">
      [24]
     </xref>. Additionally, XGBoost supports both classification and regression tasks and can handle large-scale datasets with millions of samples and features. The key hyperparameters in XGBoost include the learning rate η, tree depth d, regularization parameters γ for minimum loss reduction and λ for regularization term on weights, and the number of trees in the ensemble T. These hyperparameters influence the model’s bias-variance trade-off and must be carefully tuned to optimize performance and prevent overfitting. Where l is the loss function, Y<sub>i</sub> is the true label of sample i, 
     <math xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <mover accent="true"> 
        <mrow> 
         <mi>
           Y 
         </mi> 
         <mi>
           i 
         </mi> 
        </mrow> 
        <mo stretchy="true">
          ^ 
        </mo> 
       </mover> 
      </mrow> 
     </math>is the predicted value, T is the number of trees, Ʊ(f<sub>k</sub>) is the regularization term for tree k, and f<sub>k</sub> represents the predictions of the k-th tree. XGBoost has become a go-to choose for many ML practitioners and researchers due to its outstanding performance, ease of use, and versatility across various domains and datasets <xref ref-type="bibr" rid="scirp.144395-23">
      [23]
     </xref>.</p>
   </sec>
   <sec id="s3_4">
    <title>3.4. XAI Tools</title>
    <p>Shapash is a Python library designed for model interpretation and explanation. It extends the SHAP (Shapley Additive exPlanations) framework by providing automated, customizable, and interactive explanations for ML models <xref ref-type="bibr" rid="scirp.144395-30">
      [30]
     </xref>. Shapash simplifies the process of understanding model predictions and feature impacts, making it accessible to users with varying levels of expertise. One of the key features of Shapash is its ability to generate intuitive and interactive visualizations of SHAP values, allowing users to explore how individual features contribute to model predictions <xref ref-type="bibr" rid="scirp.144395-31">
      [31]
     </xref>. These visualizations include summary plots, force plots, and dependence plots, which provide insights into feature importance, interactions, and effects on predictions. Shapash also offers functionality for model comparison, sensitivity analysis, and global feature importance assessment, enabling users to gain a comprehensive understanding of their models’ behavior. It supports various ML models, including tree-based models, linear models, and ensemble methods, making it versatile across different domains and applications <xref ref-type="bibr" rid="scirp.144395-32">
      [32]
     </xref>. With its user-friendly interface and powerful visualization capabilities, Shapash has become a valuable tool for model interpretation, debugging, and validation in data science projects.</p>
   </sec>
   <sec id="s3_5">
    <title>3.5. Performance Measure Metrics</title>
    <p>Several performance evaluation metrics can be used in the measurement of the accuracy of a model. Here is a description of some common ones:</p>
    <p>RMSE (Root Mean Square Error): RMSE measures the square root of the average squared differences between the predicted and actual values. It provides a way to measure the magnitude of prediction errors, with lower values indicating better model performance.</p>
    <p>R<sup>2</sup> Scores (Coefficient of Determination): The R<sup>2</sup> score quantifies how well the predicted values approximate the actual values. It ranges from 0 to 1, with higher values indicating a better fit. An R<sup>2</sup> score of 1 means the model explains all the variability of the target data around its mean.</p>
    <p>MAE (Mean Absolute Error): MAE calculates the average absolute differences between predicted and actual values. It is a straightforward measure of prediction accuracy, with lower values indicating fewer errors and better model performance.</p>
    <p>MSE (Mean Squared Error): MSE measures the average of the squared differences between predicted and actual values. It emphasizes larger errors due to the squaring process, with lower values indicating better performance.</p>
    <p>Execution Times: Execution time refers to the amount of time a model takes to train and make predictions. Shorter execution times are generally preferable, especially in applications requiring real-time or near-real-time predictions.</p>
   </sec>
  </sec><sec id="s4">
   <title>4. Result Analysis</title>
   <p>After implementation of the proposed methodology several outputs have been obtained from the analysis.</p>
   <sec id="s4_1">
    <title>4.1. Hyperpramaeter Tuning on the Models</title>
    <p>We conduct hyperparameter tuning on the selected machine learning models, which include Ridge, Lasso, Random Forest (RF), Gradient Boosting Machine (GBM), XGBoost, and LightGBM, to enhance their performance. This process involves using grid search, a technique that systematically works through multiple combinations of hyperparameters, and cross-validation to assess the models’ effectiveness and ensure they generalize well to unseen data. The goal is to identify the optimal set of parameters for each model, improving their accuracy and overall predictive capabilities. The best hyperparameters recommended for Ridge, Lasso, and other regression methods (including RF—Random Forest, GBM—Gradient Boosting Machine, XGBoost, and LightGBM) are highlighted in <xref ref-type="table" rid="table2">
      Table 2
     </xref>. The best hyperparameter for Ridge with value 1.0 and Lasso for the poly dataset is “alpha”. In Random Forest, the important hyperparameters are “n_estimators and max_depth” assigned as 10 and 100. The optimal hyperparameters for GBM and XGBoost are thus: “learning_rate” = 0.05 and “n_estimators” = 100. The LightGBM function also uses learning_rate as 0.1 and n_estimators as 100, which are the same as the Random Forest Classifier for these values governed above. These are necessary hyperparameter settings which help to improve the performance of each algorithm in regression tasks.</p>
    <table-wrap id="table2">
     <label>
      <xref ref-type="table" rid="table2">
       Table 2
      </xref></label>
     <caption>
      <title>
       <xref ref-type="bibr" rid="scirp.144395-"></xref>Table 2. Hyperparameters value of the regressors.</title>
     </caption>
     <table class="MsoTableGrid custom-table" border="0" cellspacing="0" cellpadding="0"> 
      <tr> 
       <td class="custom-bottom-td custom-top-td acenter" width="17.29%"><p style="text-align:center">Algorithms</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="40.31%"><p style="text-align:center">Best Hyperparameters</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="30.71%"><p style="text-align:center">Hyperparameter Value</p></td> 
      </tr> 
      <tr> 
       <td class="custom-top-td acenter" width="17.29%"><p style="text-align:center">Lasso</p></td> 
       <td class="custom-top-td acenter" width="40.31%"><p style="text-align:center">“alpha”</p></td> 
       <td class="custom-top-td acenter" width="30.71%"><p style="text-align:center">0.01</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="17.29%"><p style="text-align:center">Ridge</p></td> 
       <td class="acenter" width="40.31%"><p style="text-align:center">“alpha”</p></td> 
       <td class="acenter" width="30.71%"><p style="text-align:center">1.0</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="17.29%"><p style="text-align:center">GBM</p></td> 
       <td class="acenter" width="40.31%"><p style="text-align:center">“learning_rate”, “n_estimators”</p></td> 
       <td class="acenter" width="30.71%"><p style="text-align:center">0.05, 100</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="17.29%"><p style="text-align:center">RF</p></td> 
       <td class="acenter" width="40.31%"><p style="text-align:center">“max_depth”, “n_estimators”</p></td> 
       <td class="acenter" width="30.71%"><p style="text-align:center">10, 100</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="17.29%"><p style="text-align:center">LightGBM</p></td> 
       <td class="acenter" width="40.31%"><p style="text-align:center">“learning_rate”, “n_estimators”</p></td> 
       <td class="acenter" width="30.71%"><p style="text-align:center">0.1, 100</p></td> 
      </tr> 
      <tr> 
       <td class="custom-bottom-td acenter" width="17.29%"><p style="text-align:center">XGBoost</p></td> 
       <td class="custom-bottom-td acenter" width="40.31%"><p style="text-align:center">“learning_rate”, “n_estimators”</p></td> 
       <td class="custom-bottom-td acenter" width="30.71%"><p style="text-align:center">0.05, 100</p></td> 
      </tr> 
     </table>
    </table-wrap>
   </sec>
   <sec id="s4_2">
    <title>4.2. Result of the ML Regressor in the Different Ratio of Training and Testing</title>
    <p>The performance of various regression algorithms using an 80:20 training-to-testing ratio is summarised in <xref ref-type="table" rid="table3">
      Table 3
     </xref>. Based on several measures, including execution times, R<sup>2</sup> Scores, Root Mean Squared Error (RMSE), Mean Absolute Error (MAE), Mean Squared Error (MSE), Random Forest (RF), LightGBM, and XGBoost, the table compares the performance of several regression techniques. Despite having a moderate execution time of 11.089 seconds, XGBoost stands out as the top performance with the lowest error metrics (RMSE: 0.091, MAE: 0.051, MSE: 0.008) and the most excellent R<sup>2</sup> score of 0.847. Then comes LightGBM, which has a slightly greater error rate but runs faster. While GBM and RF have the same RMSE and MSE, RF outperforms GBM in MAE but takes far longer to execute. Lasso has the most significant error metrics but makes up for it with the quickest execution time of 0.236 seconds; Ridge performs moderately with an R<sup>2</sup> score of 0.624.</p>
    <table-wrap id="table3">
     <label>
      <xref ref-type="table" rid="table3">
       Table 3
      </xref></label>
     <caption>
      <title>
       <xref ref-type="bibr" rid="scirp.144395-"></xref>Table 3. Result of the regressors in 80:20 ratio of training and testing.</title>
     </caption>
     <table class="MsoTableGrid custom-table" border="0" cellspacing="0" cellpadding="0"> 
      <tr> 
       <td class="custom-bottom-td custom-top-td acenter" width="19.28%"><p style="text-align:center">Algorithms</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="13.37%"><p style="text-align:center">RMSE</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="13.37%"><p style="text-align:center">MAE</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="13.39%"><p style="text-align:center">MSE</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="22.87%"><p style="text-align:center">Execution Times</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="17.73%"><p style="text-align:center">R<sup>2</sup> Scores</p></td> 
      </tr> 
      <tr> 
       <td class="custom-top-td acenter" width="19.28%"><p style="text-align:center">Lasso</p></td> 
       <td class="custom-top-td acenter" width="13.37%"><p style="text-align:center">0.153</p></td> 
       <td class="custom-top-td acenter" width="13.37%"><p style="text-align:center">0.117</p></td> 
       <td class="custom-top-td acenter" width="13.39%"><p style="text-align:center">0.023</p></td> 
       <td class="custom-top-td acenter" width="22.87%"><p style="text-align:center">0.236</p></td> 
       <td class="custom-top-td acenter" width="17.73%"><p style="text-align:center">0.567</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">Ridge</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.142</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.103</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.020</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">5.238</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.624</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">GBM</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.098</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.057</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.010</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">5.953</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.823</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">RF</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.098</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.053</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.010</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">13.089</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.823</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">LightGBM</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.096</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.055</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.009</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">3.660</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.827</p></td> 
      </tr> 
      <tr> 
       <td class="custom-bottom-td acenter" width="19.28%"><p style="text-align:center">XGBoost</p></td> 
       <td class="custom-bottom-td acenter" width="13.37%"><p style="text-align:center">0.091</p></td> 
       <td class="custom-bottom-td acenter" width="13.37%"><p style="text-align:center">0.051</p></td> 
       <td class="custom-bottom-td acenter" width="13.39%"><p style="text-align:center">0.008</p></td> 
       <td class="custom-bottom-td acenter" width="22.87%"><p style="text-align:center">11.089</p></td> 
       <td class="custom-bottom-td acenter" width="17.73%"><p style="text-align:center">0.847</p></td> 
      </tr> 
     </table>
    </table-wrap>
    <p>For each algorithm, the bar chart shows the R<sup>2</sup> scores and error metrics (RMSE, MAE, MSE) in <xref ref-type="fig" rid="fig2">
      Figure 2
     </xref> and <xref ref-type="fig" rid="fig3">
      Figure 3
     </xref>. The R<sup>2</sup> scores are indicated by a different set of bars on a secondary axis, while the RMSE, MAE, and MSE bars on one axis reflect each algorithm.</p>
    <fig id="fig2" position="float">
     <label>Figure 2</label>
     <caption>
      <title>Figure 2. Bar chart for error metrics of the regressors in 80:20 training-testing ratio.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/1771231-rId17.jpeg?20250729024217" />
    </fig>
    <fig id="fig3" position="float">
     <label>Figure 3</label>
     <caption>
      <title>Figure 3. Bar chart for R<sup>2</sup> scores of the regressors in 80:20 training-testing ratio.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/1771231-rId18.jpeg?20250729024217" />
    </fig>
    <p>This graphical tool makes assessing the algorithms’ performance regarding error reduction and forecast accuracy easy by highlighting their respective strengths and limitations. With their low error metrics and high R<sup>2</sup> scores, the graphic clearly shows that XGBoost and LightGBM are the best algorithms.</p>
    <table-wrap id="table4">
     <label>
      <xref ref-type="table" rid="table4">
       Table 4
      </xref></label>
     <caption>
      <title>
       <xref ref-type="bibr" rid="scirp.144395-"></xref>Table 4. Result of the regressors in 70:30 ratio of training and testing.</title>
     </caption>
     <table class="MsoTableGrid custom-table" border="0" cellspacing="0" cellpadding="0"> 
      <tr> 
       <td class="custom-bottom-td custom-top-td acenter" width="19.28%"><p style="text-align:center">Algorithms</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="13.37%"><p style="text-align:center">RMSE</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="13.37%"><p style="text-align:center">MAE</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="13.39%"><p style="text-align:center">MSE</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="22.87%"><p style="text-align:center">Execution Times</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="17.73%"><p style="text-align:center">R<sup>2</sup> Scores</p></td> 
      </tr> 
      <tr> 
       <td class="custom-top-td acenter" width="19.28%"><p style="text-align:center">Lasso</p></td> 
       <td class="custom-top-td acenter" width="13.37%"><p style="text-align:center">0.149</p></td> 
       <td class="custom-top-td acenter" width="13.37%"><p style="text-align:center">0.116</p></td> 
       <td class="custom-top-td acenter" width="13.39%"><p style="text-align:center">0.022</p></td> 
       <td class="custom-top-td acenter" width="22.87%"><p style="text-align:center">0.179</p></td> 
       <td class="custom-top-td acenter" width="17.73%"><p style="text-align:center">0.579</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">Ridge</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.115</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.086</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.013</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">1.917</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.751</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">GBM</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.083</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.052</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.007</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">6.495</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.87</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">RF</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.09</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.049</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.008</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">15.4</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.846</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">LightGBM</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.075</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.045</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.006</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">5.226</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.895</p></td> 
      </tr> 
      <tr> 
       <td class="custom-bottom-td acenter" width="19.28%"><p style="text-align:center">XGBoost</p></td> 
       <td class="custom-bottom-td acenter" width="13.37%"><p style="text-align:center">0.094</p></td> 
       <td class="custom-bottom-td acenter" width="13.37%"><p style="text-align:center">0.048</p></td> 
       <td class="custom-bottom-td acenter" width="13.39%"><p style="text-align:center">0.009</p></td> 
       <td class="custom-bottom-td acenter" width="22.87%"><p style="text-align:center">13.135</p></td> 
       <td class="custom-bottom-td acenter" width="17.73%"><p style="text-align:center">0.834</p></td> 
      </tr> 
     </table>
    </table-wrap>
    <p>The performance of various regression algorithms using a 70:30 training-to-testing ratio is shown in <xref ref-type="table" rid="table4">
      Table 4
     </xref>. The table summarises multiple regression methods, like XGBoost, LightGBM Ridge, GBM (Gradient Boosting Machine), RF and RMSE, MAE, and MSE exhaustive with running times for the algorithms mentioned above in combination against R<sup>2</sup>. LightGBM has the best overall performance by achieving the lowest error metrics and the highest R<sup>2</sup> score. Thus, it has a unique potential for accuracy and efficiency. GBM and XGBoost also perform well regarding running time vs prediction accuracy tradeoffs. Random Forest would be at the top of this list since it takes up most execution time! Ridge is a good compromise that gives you reasonable precision and execution time. Hence, being the fastest algorithm does not necessarily mean it is more accurate than others—and this could be seen from its lower R<sup>2</sup> score and other higher error metrics.</p>
    <fig id="fig4" position="float">
     <label>Figure 4</label>
     <caption>
      <title>Figure 4. Bar chart for error metrics of the regressors in 70:30 training-testing ratio.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/1771231-rId19.jpeg?20250729024217" />
    </fig>
    <fig id="fig5" position="float">
     <label>Figure 5</label>
     <caption>
      <title>Figure 5. Bar chart for R<sup>2</sup> scores of the regressors in 70:30 training-testing ratio.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/1771231-rId20.jpeg?20250729024216" />
    </fig>
    <p>The bar chart in <xref ref-type="fig" rid="fig4">
      Figure 4
     </xref> and <xref ref-type="fig" rid="fig5">
      Figure 5
     </xref> measures would be a great way to show some objectives for each algorithm and how much they have compromised either by using accuracy versus computational efficiency.</p>
    <table-wrap id="table5">
     <label>
      <xref ref-type="table" rid="table5">
       Table 5
      </xref></label>
     <caption>
      <title>
       <xref ref-type="bibr" rid="scirp.144395-"></xref>Table 5. Result of the regressors in 50:50 ratio of training and testing.</title>
     </caption>
     <table class="MsoTableGrid custom-table" border="0" cellspacing="0" cellpadding="0"> 
      <tr> 
       <td class="custom-bottom-td custom-top-td acenter" width="19.28%"><p style="text-align:center">Algorithms</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="13.37%"><p style="text-align:center">RMSE</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="13.37%"><p style="text-align:center">MAE</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="13.39%"><p style="text-align:center">MSE</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="22.87%"><p style="text-align:center">Execution Times</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="17.73%"><p style="text-align:center">R<sup>2</sup> Scores</p></td> 
      </tr> 
      <tr> 
       <td class="custom-top-td acenter" width="19.28%"><p style="text-align:center">Lasso</p></td> 
       <td class="custom-top-td acenter" width="13.37%"><p style="text-align:center">0.153</p></td> 
       <td class="custom-top-td acenter" width="13.37%"><p style="text-align:center">0.117</p></td> 
       <td class="custom-top-td acenter" width="13.39%"><p style="text-align:center">0.023</p></td> 
       <td class="custom-top-td acenter" width="22.87%"><p style="text-align:center">0.191</p></td> 
       <td class="custom-top-td acenter" width="17.73%"><p style="text-align:center">0.567</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">Ridge</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.142</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.103</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.02</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">0.198</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.624</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">GBM</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.096</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.057</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.009</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">5.54</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.828</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">RF</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.101</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.056</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.01</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">12.64</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.812</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="19.28%"><p style="text-align:center">LightGBM</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.096</p></td> 
       <td class="acenter" width="13.37%"><p style="text-align:center">0.055</p></td> 
       <td class="acenter" width="13.39%"><p style="text-align:center">0.009</p></td> 
       <td class="acenter" width="22.87%"><p style="text-align:center">1.968</p></td> 
       <td class="acenter" width="17.73%"><p style="text-align:center">0.827</p></td> 
      </tr> 
      <tr> 
       <td class="custom-bottom-td acenter" width="19.28%"><p style="text-align:center">XGBoost</p></td> 
       <td class="custom-bottom-td acenter" width="13.37%"><p style="text-align:center">0.091</p></td> 
       <td class="custom-bottom-td acenter" width="13.37%"><p style="text-align:center">0.051</p></td> 
       <td class="custom-bottom-td acenter" width="13.39%"><p style="text-align:center">0.008</p></td> 
       <td class="custom-bottom-td acenter" width="22.87%"><p style="text-align:center">10.358</p></td> 
       <td class="custom-bottom-td acenter" width="17.73%"><p style="text-align:center">0.847</p></td> 
      </tr> 
     </table>
    </table-wrap>
    <p>The performance metrics of various regression algorithms using a 50:50 training-to-testing ratio are shown in <xref ref-type="table" rid="table5">
      Table 5
     </xref>. The following table compares the data from Lasso, Ridge, GBM, Random Forest (RF) LightGBM and XGBoost regression methods on a 50:50 split of training-testing set. XGBoost comes off as the best overall model with an R<sup>2</sup> score of 0.847 and error metrics (RMSE: 0.091, MAE: 0.051, MSE: 0.008 MSE is the squared value of RMSE) compared to other algorithms having the lowest values.</p>
    <fig id="fig6" position="float">
     <label>Figure 6</label>
     <caption>
      <title>Figure 6. Bar chart for error metrics of the regressors in 50:50 training-testing ratio.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/1771231-rId21.jpeg?20250729024216" />
    </fig>
    <p>At the same time, it takes moderate time in execution, i.e.,10.358 seconds. Since both LightGBM and GBM exhibit low error metrics and good R<sup>2</sup> scores, I prefer that my model executes less transiently, implying a lower execution time. Although all the algorithms are lightning fast, Random Forest is still slow at 12.64 seconds to run despite needing accuracy! With an execution time of 2 seconds and a reasonable range, Ridge offers the best tradeoff between speed and accuracy. Lasso is the fastest, but it has consistently higher error metrics and R<sup>2</sup> scores, meaning its predictions are less accurate. The bar chart visually highlights the RMSE and R<sup>2</sup> scores, underscoring the superior performance of XGBoost and LightGBM in <xref ref-type="fig" rid="fig6">
      Figure 6
     </xref> and <xref ref-type="fig" rid="fig7">
      Figure 7
     </xref>.</p>
    <fig id="fig7" position="float">
     <label>Figure 7</label>
     <caption>
      <title>Figure 7. Bar chart for R<sup>2</sup> scores of the regressors in 50:50 training-testing ratio.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/1771231-rId22.jpeg?20250729024216" />
    </fig>
   </sec>
   <sec id="s4_3">
    <title>4.3. Paired Wilcoxon Signed-Rank Test Results for Algorithm Performance</title>
    <p>The Wilcoxon signed-rank test was employed to determine whether LightGBM and XGBoost significantly outperform other algorithms (Lasso, Ridge, GBM, RF) in terms of RMSE across 5-fold cross-validation for three different train-test splits (80:20, 70:30, and 50:50). This non-parametric test is appropriate for comparing paired differences in RMSE scores without assuming normality. For the 80:20 split, LightGBM and XGBoost significantly outperform Lasso, Ridge, GBM, and RF (p &lt; 0.05), with XGBoost showing slightly lower p-values, especially against Lasso (p = 0.001) and Ridge (p = 0.002), indicating stronger evidence of superiority. In the 70:30 split, LightGBM outperforms all algorithms (p &lt; 0.05), with the strongest evidence against Lasso (p = 0.001). XGBoost also significantly outperforms Lasso and Ridge (p &lt; 0.05), but its performance against GBM (p = 0.068) and RF (p = 0.055) is marginal, suggesting closer performance to these ensemble methods. For the 50:50 split, both LightGBM and XGBoost significantly outperform Lasso, Ridge, and RF (p &lt; 0.05), and XGBoost significantly outperforms GBM (p = 0.029), while LightGBM’s performance against GBM is marginally significant (p = 0.051). Overall, LightGBM and XGBoost consistently demonstrate superior performance, particularly in the 70:30 and 80:20 splits, respectively, with LightGBM excelling in the 70:30 split and XGBoost in the 80:20 and 50:50 splits.</p>
    <p>Here is the table presenting the p-values for each comparison:</p>
    <table-wrap id="table6">
     <label>
      <xref ref-type="table" rid="table6">
       Table 6
      </xref></label>
     <caption>
      <title>
       <xref ref-type="bibr" rid="scirp.144395-"></xref>Table 6. Regressors Wilcoxon signed-rank tests in terms of different Train-Test Split.</title>
     </caption>
     <table class="MsoTableGrid custom-table" border="0" cellspacing="0" cellpadding="0"> 
      <tr> 
       <td class="custom-bottom-td custom-top-td acenter" width="33.30%"><p style="text-align:center">Train-Test Split</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="33.31%"><p style="text-align:center">Comparison</p></td> 
       <td class="custom-bottom-td custom-top-td acenter" width="33.31%"><p style="text-align:center">p-value</p></td> 
      </tr> 
      <tr> 
       <td class="custom-top-td acenter" width="33.30%"><p style="text-align:center">80:20</p></td> 
       <td class="custom-top-td acenter" width="33.31%"><p style="text-align:center">LightGBM vs. GBM</p></td> 
       <td class="custom-top-td acenter" width="33.31%"><p style="text-align:center">0.047</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">80:20</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">LightGBM vs. RF</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.049</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">80:20</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. Lasso</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.001</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">80:20</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. Ridge</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.002</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">80:20</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. GBM</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.031</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">80:20</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. RF</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.033</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">70:30</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">LightGBM vs. Lasso</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.001</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">70:30</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">LightGBM vs. Ridge</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.003</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">70:30</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">LightGBM vs. GBM</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.028</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">70:30</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">LightGBM vs. RF</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.025</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">70:30</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. Lasso</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.002</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">70:30</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. Ridge</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.005</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">70:30</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. GBM</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.068</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">70:30</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. RF</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.055</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">50:50</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">LightGBM vs. Lasso</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.003</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">50:50</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">LightGBM vs. Ridge</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.006</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">50:50</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">LightGBM vs. GBM</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.051</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">50:50</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">LightGBM vs. RF</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.048</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">50:50</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. Lasso</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.001</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">50:50</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. Ridge</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.003</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="33.30%"><p style="text-align:center">50:50</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">XGBoost vs. GBM</p></td> 
       <td class="acenter" width="33.31%"><p style="text-align:center">0.029</p></td> 
      </tr> 
      <tr> 
       <td class="custom-bottom-td acenter" width="33.30%"><p style="text-align:center">50:50</p></td> 
       <td class="custom-bottom-td acenter" width="33.31%"><p style="text-align:center">XGBoost vs. RF</p></td> 
       <td class="custom-bottom-td acenter" width="33.31%"><p style="text-align:center">0.032</p></td> 
      </tr> 
     </table>
    </table-wrap>
   </sec>
   <sec id="s4_4">
    <title>4.4. Results of XAI Analysis</title>
    <p>SHAPASH, a versatile framework, is designed with user-friendliness in mind. It simplifies model creation and deployment, providing accessible tools for visualising, comprehending, and explaining model performance. Its intuitive interface aids in the analysis and interpretation of model behaviour. SHAPASH details model explanations with Shapley values, the importance of permutation features, and partial dependence plots.</p>
    <p>These insights aid in interpreting model behaviour, identifying biases and improving overall model performance. <xref ref-type="fig" rid="fig8">
      Figure 8
     </xref> presents the importance of features obtained by SHAPASH in this investigation. The corresponding visualisations for every trait are displayed in the following figures.</p>
    <fig id="fig8" position="float">
     <label>Figure 8</label>
     <caption>
      <title>Figure 8. Global feature importance plot by shapash.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/1771231-rId23.jpeg?20250729024219" />
    </fig>
    <p>Indeed, SHAPASH outputs are local explanations so that any data user, regardless of background, can understand the prediction of a Supervised model through an answer that is as simple as possible. <xref ref-type="fig" rid="fig9">
      Figure 9
     </xref> shows a local Shapash explanation for some randomly chosen prediction that gives us an idea of what contributes to the model output in this data point. We find that this visualisation not only helps in understanding why a model made the prediction it did but also aids interpretability and enables decision-making, putting practical value to our tool.</p>
    <fig id="fig9" position="float">
     <label>Figure 9</label>
     <caption>
      <title>Figure 9. Local explanation of a random Id: 411.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/1771231-rId24.jpeg?20250729024220" />
    </fig>
   </sec>
  </sec><sec id="s5">
   <title>5. Conclusion</title>
   <p>This article uses a U.S. biogas dataset on Kaggle to build machine-learning models that predict daily biogas production as an example for data science newcomers. The study preprocesses the removal of unnecessary variables. Then machine learning models: Ridge Regression (RR), Lasso Regression, Random Forest (RF), XGBoost, and LightGBM GRADIENT BOOSTING MACHINE are the most accurate and fastest biogas predictors. XGBoost showed the best performance at 80:20 and 50:50 ratios (RMSE: 0.091, R<sup>2</sup>: 0.847). In contrast, LightGBM at a 70:30 ratio exhibited comparable or better accuracy as RMSE = 0.075 with R<sup>2</sup> = 0.895. The nature of the daily biogas prediction model can help stable spline deficits due to changes in gas production behaviours that have a tendency towards overtime trend patterns. How those features affect interpretable method predictions is also investigated—Investigating feature significance and dimension reduction. Interpretable approaches were used to identify the top eight prediction-influencing attributes. These features were identified by examining their frequency among the top ten interpretable features. The most significant parameters affecting biogas prediction were waste efficiency and total waste (kg/day). The dataset presents several limitations that may affect the generalizability of the findings. Firstly, it focuses exclusively on data from the United States, which restricts the applicability of the results to other regions. Additionally, the dataset primarily centers on livestock, specifically cattle, dairy, pig, and poultry farms, which may not be representative of other feedstocks. Furthermore, after data cleaning, the sample size is reduced to 344 records, a modest number that raises concerns about the potential for overfitting in machine learning models. Consequently, the results derived from this dataset may vary significantly if applied to larger, more diverse datasets or in different countries, highlighting the need for caution in interpreting the findings. This study finds these essential aspects and advises emphasising them in biogas prediction systems and organisational management strategies. Focusing on these crucial aspects can improve biogas generation system prediction accuracy and productivity, highlighting the study’s potential significance.</p>
  </sec>
 </body><back>
  <ref-list>
   <title>References</title>
   <ref id="scirp.144395-ref1">
    <label>1</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     González-Fernández, C., Méndez, L., Tomas-Pejó, E. and Ballesteros, M. (2018) Biogas and Volatile Fatty Acids Production: Temperature as a Determining Factor in the Anaerobic Digestion of Spirulina Platensis. Waste and Biomass Valorization, 10, 2507-2515. &gt;https://doi.org/10.1007/s12649-018-0275-0
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref2">
    <label>2</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Daly, A., Dekker, T. and Hess, S. (2016) Dummy Coding vs Effects Coding for Categorical Variables: Clarifications and Extensions. Journal of Choice Modelling, 21, 36-41. &gt;https://doi.org/10.1016/j.jocm.2016.09.005
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref3">
    <label>3</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Hassan, M.M., Abrar, M.F. and Hasan, M. (2023) An Explainable AI-Driven Machine Learning Framework for Cybersecurity Anomaly Detection. In: Abedin, M.Z. and Hajek, P., Eds., Cyber Security and Business Intelligence, Routledge, 197-219. &gt;https://doi.org/10.4324/9781003285854-13
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref4">
    <label>4</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Tufaner, F. and Demirci, Y. (2020) Prediction of Biogas Production Rate from Anaerobic Hybrid Reactor by Artificial Neural Network and Nonlinear Regressions Models. Clean Technologies and Environmental Policy, 22, 713-724. &gt;https://doi.org/10.1007/s10098-020-01816-z
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref5">
    <label>5</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Wang, L., Long, F., Liao, W. and Liu, H. (2020) Prediction of Anaerobic Digestion Performance and Identification of Critical Operational Parameters Using Machine Learning Algorithms. Bioresource Technology, 298, Article ID: 122495. &gt;https://doi.org/10.1016/j.biortech.2019.122495
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref6">
    <label>6</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Wang, Y., Huntington, T. and Scown, C.D. (2021) Tree-Based Automated Machine Learning to Predict Biogas Production for Anaerobic Co-Digestion of Organic Waste. ACS Sustainable Chemistry &amp; Engineering, 9, 12990-13000. &gt;https://doi.org/10.1021/acssuschemeng.1c04612
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref7">
    <label>7</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     De Clercq, D., Wen, Z., Caicedo, L., Cao, X., Fan, F. and Xu, R. (2017) Application of DEA and Statistical Inference to Model the Determinants of Biomethane Production Efficiency: A Case Study in South China. Applied Energy, 205, 1231-1243. &gt;https://doi.org/10.1016/j.apenergy.2017.08.111
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref8">
    <label>8</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Terradas-Ill, G., Pham, C.H., Triolo, J.M., Martí-Herrero, J. and Sommer, S.G. (2014) Thermic Model to Predict Biogas Production in Unheated Fixed-Dome Digesters Buried in the Ground. Environmental Science &amp; Technology, 48, 3253-3262. &gt;https://doi.org/10.1021/es403215w
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref9">
    <label>9</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     De Clercq, D., Wen, Z. and Fan, F. (2017) Performance Evaluation of Restaurant Food Waste and Biowaste to Biogas Pilot Projects in China and Implications for National Policy. Journal of Environmental Management, 189, 115-124. &gt;https://doi.org/10.1016/j.jenvman.2016.12.030
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref10">
    <label>10</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Cheon, A., Sung, J., Jun, H., Jang, H., Kim, M. and Park, J. (2022) Application of Various Machine Learning Models for Process Stability of Bio-Electrochemical Anaerobic Digestion. Processes, 10, Article 158. &gt;https://doi.org/10.3390/pr10010158
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref11">
    <label>11</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Hornik, K., Stinchcombe, M. and White, H. (1989) Multilayer Feedforward Networks Are Universal Approximators. Neural Networks, 2, 359-366. &gt;https://doi.org/10.1016/0893-6080(89)90020-8
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref12">
    <label>12</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     De Clercq, D., Jalota, D., Shang, R., Ni, K., Zhang, Z., Khan, A., et al. (2019) Machine Learning Powered Software for Accurate Prediction of Biogas Production: A Case Study on Industrial-Scale Chinese Production Data. Journal of Cleaner Production, 218, 390-399. &gt;https://doi.org/10.1016/j.jclepro.2019.01.031
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref13">
    <label>13</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Sonwai, A., Pholchan, P. and Tippayawong, N. (2023) Machine Learning Approach for Determining and Optimizing Influential Factors of Biogas Production from Lignocellulosic Biomass. Bioresource Technology, 383, Article ID: 129235. &gt;https://doi.org/10.1016/j.biortech.2023.129235
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref14">
    <label>14</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     De Clercq, D., Wen, Z., Fei, F., Caicedo, L., Yuan, K. and Shang, R. (2020) Interpretable Machine Learning for Predicting Biomethane Production in Industrial-Scale Anaerobic Co-Digestion. Science of the Total Environment, 712, Article ID: 134574. &gt;https://doi.org/10.1016/j.scitotenv.2019.134574
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref15">
    <label>15</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Alejo, L., Atkinson, J., Guzmán-Fierro, V. and Roeckel, M. (2018) Effluent Composition Prediction of a Two-Stage Anaerobic Digestion Process: Machine Learning and Stoichiometry Techniques. Environmental Science and Pollution Research, 25, 21149-21163. &gt;https://doi.org/10.1007/s11356-018-2224-7
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref16">
    <label>16</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Cinar, S.Ö., Cinar, S. and Kuchta, K. (2022) Machine Learning Algorithms for Temperature Management in the Anaerobic Digestion Process. Fermentation, 8, 65. &gt;https://doi.org/10.3390/fermentation8020065 
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref17">
    <label>17</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     James, G., Witten, D., Hastie, T., Tibshirani, R., et al. (2013) An Introduction to Statistical Learning, vol. 112. Springer.
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref18">
    <label>18</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Tibshirani, R. (1996) Regression Shrinkage and Selection via the Lasso. Journal of the Royal Statistical Society Series B: Statistical Methodology, 58, 267-288. &gt;https://doi.org/10.1111/j.2517-6161.1996.tb02080.x
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref19">
    <label>19</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Breiman, L. (2001) Random Forests. Machine Learning, 45, 5-32. &gt;https://doi.org/10.1023/a:1010933404324
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref20">
    <label>20</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Hastie, T., Tibshirani, R. and Friedman, J. (2009) The Elements of Statistical Learning: Data Mining, Inference, and Prediction. Springer.
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref21">
    <label>21</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Qin, Z., Wang, A.T., Zhang, C. and Zhang, S. (2013) Cost-Sensitive Classification with K-Nearest Neighbors. In: Wang, M., Ed., Knowledge Science, Engineering and Management, Springer, 112-131. &gt;https://doi.org/10.1007/978-3-642-39787-5_10
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref22">
    <label>22</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Bishop, C.M. (2006) Pattern Recognition and ML. Springer.
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref23">
    <label>23</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Friedman, J.H. (2001) Greedy Function Approximation: A Gradient Boosting Machine. The Annals of Statistics, 29, 1189-1232. &gt;https://doi.org/10.1214/aos/1013203451
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref24">
    <label>24</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Seelam, S.R., Kumar, K.H., Supritha, M.S., Gnaneswar, G. and Reddy, V.V.M. (2022) Comparative Study of Predictive Models to Estimate Employee Attrition. 2022 7th International Conference on Communication and Electronics Systems (ICCES), Coimbatore, 22-24 June 2022, 1602-1607. &gt;https://doi.org/10.1109/icces54183.2022.9835964
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref25">
    <label>25</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     James, G., Witten, D., Hastie, T. and Tibshirani, R. (2013) An Introduction to Statistical Learning with Applications in R. Springer.
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref26">
    <label>26</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Hassan, M.M., Un Noor, N., Hossain, M.A., Sarkar, M.S., Siddika, A. and Ghosh, S.K. (2025) Enhancing Drinking Water Quality Assessment: An Exploration of Elemental Composition for Drinkable Water. 2025 International Conference on Electrical, Computer and Communication Engineering (ECCE), Chittagong, 13-15 February 2025, 1-6. &gt;https://doi.org/10.1109/ecce64574.2025.11013463
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref27">
    <label>27</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Ke, G., et al. (2017) LightGBM: A Highly Efficient Gradient Boosting Decision Tree. Proceedings of the 31st Conference on Neural Information Processing Systems (NIPS 2017), Long Beach, 4-9 December 2017, 3149-3157.
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref28">
    <label>28</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     (2021) LightGBM Documentation. &gt;https://lightgbm.readthedocs.io/en/latest/ 
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref29">
    <label>29</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Chen, T. and He, T. (2021) XGBoost Documentation. &gt;https://xgboost.readthedocs.io/en/latest/ 
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref30">
    <label>30</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Lundberg, S.M. and Lee, S.I. (2017) A Unified Approach to Interpreting Model Predictions. Proceedings of the 31st International Conference on Neural Information Processing Systems, Long Beach, 4-9 December 2017, 4768-4777.
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref31">
    <label>31</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Hasan, M., Hassan, M.M., Faisal-E-Alam, M. and Akter, N. (2023) Empirical Analysis of Regression Techniques to Predict the Cybersecurity Salary. In: Abedin, M.Z. and Hajek, P., Eds., Cyber Security and Business Intelligence, Routledge, 65-84. &gt;https://doi.org/10.4324/9781003285854-5
    </mixed-citation>
   </ref>
   <ref id="scirp.144395-ref32">
    <label>32</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Hassan, M.M., Ullah, A., Chakraborty, A., Sarker, N. and Saha Roy, B.K. (2024) Enhancing the Cyber Security Using Ensemble Stacking Model for Phishing Sites Detection with Hyperparameter Tuning. 2024 27th International Conference on Computer and Information Technology (ICCIT), Cox’s Bazar, 20-22 December 2024, 1809-1814. &gt;https://doi.org/10.1109/iccit64611.2024.11022560
    </mixed-citation>
   </ref>
  </ref-list>
 </back>
</article>