<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMES</journal-id>
<journal-id journal-id-type="nlm-ta">CMES</journal-id>
<journal-id journal-id-type="publisher-id">CMES</journal-id>
<journal-title-group>
<journal-title>Computer Modeling in Engineering &#x0026; Sciences</journal-title>
</journal-title-group>
<issn pub-type="epub">1526-1506</issn>
<issn pub-type="ppub">1526-1492</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">67412</article-id>
<article-id pub-id-type="doi">10.32604/cmes.2025.067412</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>ELM-APDPs: An Explainable Ensemble Learning Method for Accurate Prediction of Druggable Proteins</article-title>
<alt-title alt-title-type="left-running-head">ELM-APDPs: An Explainable Ensemble Learning Method for Accurate Prediction of Druggable Proteins</alt-title>
<alt-title alt-title-type="right-running-head">ELM-APDPs: An Explainable Ensemble Learning Method for Accurate Prediction of Druggable Proteins</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Rehman</surname><given-names>Mujeebu</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Liu</surname><given-names>Qinghua</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Ghulam</surname><given-names>Ali</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Ahmad</surname><given-names>Tariq</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-5" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Khan</surname><given-names>Jawad</given-names></name><xref ref-type="aff" rid="aff-4">4</xref><email>jkhanbk1@gachon.ac.kr</email></contrib>
<contrib id="author-6" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Hussain</surname><given-names>Dildar</given-names></name><xref ref-type="aff" rid="aff-5">5</xref><email>hussain.bangash@sejong.ac.kr</email></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Gu</surname><given-names>Yeong Hyeon</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<aff id="aff-1"><label>1</label><institution>School of Information and Communication Engineering, Guilin University of Electronic Technology</institution>, <addr-line>Guilin, 541004</addr-line>, <country>China</country></aff>
<aff id="aff-2"><label>2</label><institution>Information Technology Centre, Sindh Agriculture University</institution>, <addr-line>Tandojam, 70060</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-3"><label>3</label><institution>School of Electrical and Information Engineering, Hunan University</institution>, <addr-line>Changsha, 410082</addr-line>, <country>China</country></aff>
<aff id="aff-4"><label>4</label><institution>School of Computing, Gachon University</institution>, <addr-line>Seongnam, 13120</addr-line>, <country>Republic of Korea</country></aff>
<aff id="aff-5"><label>5</label><institution>Department AI and Data Science, Sejong University</institution>, <addr-line>Seoul, 05006</addr-line>, <country>Republic of Korea</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Authors: Jawad Khan. Email: <email>jkhanbk1@gachon.ac.kr</email>; Dildar Hussain. Email: <email>hussain.bangash@sejong.ac.kr</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2025</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>30</day><month>10</month><year>2025</year>
</pub-date>
<volume>145</volume>
<issue>1</issue>
<fpage>779</fpage>
<lpage>805</lpage>
<history>
<date date-type="received">
<day>02</day>
<month>5</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>9</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 The Authors.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMES_67412.pdf"></self-uri>
<abstract>
<p>Identifying druggable proteins, which are capable of binding therapeutic compounds, remains a critical and resource-intensive challenge in drug discovery. To address this, we propose CEL-IDP (Comparison of Ensemble Learning Methods for Identification of Druggable Proteins), a computational framework combining three feature extraction methods Dipeptide Deviation from Expected Mean (DDE), Enhanced Amino Acid Composition (EAAC), and Enhanced Grouped Amino Acid Composition (EGAAC) with ensemble learning strategies (Bagging, Boosting, Stacking) to classify druggable proteins from sequence data. DDE captures dipeptide frequency deviations, EAAC encodes positional amino acid information, and EGAAC groups residues by physicochemical properties to generate discriminative feature vectors. These features were analyzed using ensemble models to overcome the limitations of single classifiers. EGAAC outperformed DDE and EAAC, with Random Forest (Bagging) and XGBoost (Boosting) achieving the highest accuracy of 71.66%, demonstrating superior performance in capturing critical biochemical patterns. Stacking showed intermediate results (68.33%), while EAAC and DDE-based models yielded lower accuracies (56.66%&#x2013;66.87%). CEL-IDP streamlines large-scale druggability prediction, reduces reliance on costly experimental screening, and aligns with global initiatives like Target 2035 to expand action-able drug targets. This work advances machine learning-driven drug discovery by systematizing feature engineering and ensemble model optimization, providing a scalable workflow to accelerate target identification and validation.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Druggable proteins</kwd>
<kwd>ensemble learning</kwd>
<kwd>computational drug discovery</kwd>
<kwd>pharmacological target identification</kwd>
<kwd>machine learning</kwd>
<kwd>feature extraction</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>Information Technology Research Centre</funding-source>
<award-id>IITP-2024-RS-2024-00437191</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>The human genome is made up of approximately 20,000 genes that encode proteins. However, it should be noted that not all proteins are viable targets for drug development [<xref ref-type="bibr" rid="ref-1">1</xref>&#x2013;<xref ref-type="bibr" rid="ref-3">3</xref>]. The precise identification of therapeutic targets within the human body remains critical for the development of innovative pharmaceuticals. Compared to conventional experimental techniques, machine learning (ML)-based approaches have gained increasing attention due to their efficiency and predictive accuracy in drug-target identification tasks [<xref ref-type="bibr" rid="ref-4">4</xref>,<xref ref-type="bibr" rid="ref-5">5</xref>]. Recent advancements in explainable artificial intelligence have further strengthened ML-based models in biomedical applications, particularly for druggable protein prediction tasks [<xref ref-type="bibr" rid="ref-6">6</xref>]. A druggable protein is a protein that has the ability to strongly attach to tiny drug-like compounds and has beneficial therapeutic effects [<xref ref-type="bibr" rid="ref-7">7</xref>,<xref ref-type="bibr" rid="ref-8">8</xref>]. Druggable proteins typically belong to extensive protein families that have been effectively recognised as targets for drug development [<xref ref-type="bibr" rid="ref-9">9</xref>]. The primary cause of project failures in the field of drug development is commonly attributed to the undruggable nature of the target, as indicated by an estimated 60% of all cases [<xref ref-type="bibr" rid="ref-10">10</xref>]. Thus, the druggability of a protein plays a critical role in the advancement of a drug development initiative, as it is imperative to identify drug targets [<xref ref-type="bibr" rid="ref-11">11</xref>] accurately.</p>
<p>Computational techniques that rely exclusively on the primary sequences of pharmaceuticals have the potential to enhance experimental approaches by accelerating the process of characterizing and predicting proteins that are amenable to drug development. This urgency is amplified due to the extensive production of new proteins by next-generation sequencing, which presents a significant opportunity to uncover potential druggable proteins that have not yet been described [<xref ref-type="bibr" rid="ref-12">12</xref>]. Conventional experimental techniques can accurately detect the drug targets; however, these techniques are time-consuming and difficult for applications on a large scale [<xref ref-type="bibr" rid="ref-13">13</xref>]. For instance, the investigation of the three-dimensional structure of a protein is a necessary component of experimental procedures, leading to a protracted development cycle [<xref ref-type="bibr" rid="ref-14">14</xref>]. In contrast, computational methods that depend exclusively on pharmaceutical primary sequences can complement experiments to prioritize candidates efficiently [<xref ref-type="bibr" rid="ref-15">15</xref>].</p>
<p>Recent studies emphasize key criteria for optimal drug targets. Gashaw et al. note that an optimal drug target must possess selective expression in anatomical regions, minimal physiological impact, and compatibility with high-throughput screening [<xref ref-type="bibr" rid="ref-16">16</xref>,<xref ref-type="bibr" rid="ref-17">17</xref>]. Similarly, tools like DoGSiteScorer identify druggable binding pockets using geometric and physicochemical features [<xref ref-type="bibr" rid="ref-18">18</xref>], while sequence-based ML models like DrugMiner and DrugFinder [<xref ref-type="bibr" rid="ref-19">19</xref>], XGB-DrugPred [<xref ref-type="bibr" rid="ref-20">20</xref>], and others leverage amino acid composition for predictions [<xref ref-type="bibr" rid="ref-21">21</xref>,<xref ref-type="bibr" rid="ref-22">22</xref>]. Despite progress, fewer than 20% of the &#x007E;3000 proteins in the druggable genome are targeted by FDA-approved drugs [<xref ref-type="bibr" rid="ref-23">23</xref>], underscoring the need for improved frameworks.</p>
<p>Our hypothesis posits that the WDR protein family contains a greater number of druggable members [<xref ref-type="bibr" rid="ref-24">24</xref>&#x2013;<xref ref-type="bibr" rid="ref-26">26</xref>]. We aimed to investigate this hypothesis within the framework of Target 2035, a worldwide endeavor to create pharmacological interventions for every human protein to address scalability. We focus on DEL-ML, a method overcoming practical limitations of DNA-encoded library screening [<xref ref-type="bibr" rid="ref-27">27</xref>,<xref ref-type="bibr" rid="ref-28">28</xref>]. We evaluate three feature extraction strategies: Dipeptide Deviation from Expected Mean (DDE), Enhanced Amino Acid Composition (EAAC), and Enhanced Grouped Amino Acid Composition (EGAAC) to identify optimal sequence descriptors for druggability prediction.</p>
<p>EGAAC, a feature extraction method grouping amino acids by physicochemical properties (e.g., hydrophobicity, charge), outperformed other techniques in capturing critical biochemical patterns for druggability prediction. Our approach employs ensemble learning, a technique combining multiple base models to enhance prediction accuracy. For example, the Gradient Boosting, Extreme Gradient Boosting, and Cat Boosting models achieved accuracies up to 68.33%, while Extreme Gradient Boosting Machine (XGBM) and Stacking models yielded 71.66% accuracy in dataset 1. These results demonstrate that ensemble methods outperform single classifiers, with EGAAC-based models showing superior performance.</p>
<p>We introduce CEL-IDP (Comparison of Ensemble Learning Methods for Identification of Druggable Proteins), a computational framework integrating amino acid composition features with stacking, boosting, and bagging strategies. By leveraging sequence-derived descriptors from protein primary structures, CEL-IDP captures informative biochemical patterns to prioritize druggable targets [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-30">30</xref>]. This work advances machine learning-driven drug discovery by systematizing ensemble learning for high-throughput druggability prediction, addressing the need for scalable methods to analyze large protein datasets, prioritizing the understudied WDR family, and contributing toward global efforts like Target 2035 [<xref ref-type="bibr" rid="ref-31">31</xref>].</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Benchmark Druggable Protein Dataset (Evidence)</title>
<p>The benchmark dataset typically comprises positive samples (proteins capable of interacting with drugs) and negative samples (proteins incapable of interacting with drugs). We used Jamali et al.&#x2019;s dataset [<xref ref-type="bibr" rid="ref-32">32</xref>]. To ensure a reliable comparison with current approaches. The 1611 druggable proteins have been obtained by using the Drug-Bank database, as previously reported. Similar sequences of these proteins, with respect to characteristics and composition, were eliminated using the Cluster Database at High Identity with Tolerance (CD-HIT) tool. Additionally, there are 1224 druggable proteins in the final set of positive samples. Likewise, the negative sample set was created by integrating the datasets presented by Bakheet and Doig [<xref ref-type="bibr" rid="ref-33">33</xref>]. First, the Swiss-Prot repository was used to obtain these sequences. Following the elimination of analogous sequences, 1300 non-druggable proteins persisted. The final benchmark dataset, which consists of 1218 druggable proteins, is employed in experimental data points amounting to 2518 druggable proteins, as shown in <xref ref-type="table" rid="table-1">Table 1</xref>.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Druggable and non-druggable proteins in the benchmark and independent datasets</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th></th>
<th>Benchmarking</th>
<th>Independent</th>
</tr>
</thead>
<tbody>
<tr>
<td>Druggable proteins</td>
<td>1224</td>
<td>1218</td>
</tr>
<tr>
<td>Non-Druggable proteins</td>
<td>1611</td>
<td>1300</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Feature Extraction Using Dipeptide Deviation from Expected Mean (DDE)</title>
<p>The normalization method is implemented on the data of each DDE vector feature profile, which comprises 400 elements. For the 400 vector characteristics extracted from the original protein profiles, the DDE model demonstrates the generation of score matrix functions. The initial step involves synchronizing all values with a common amino value. The frequency of amino acids is subsequently divided by the length of the sequence. In conclusion, formula X is used to scale all functional values. Means and standard deviations are derived using vector score feature profiles, which consist of 400 vector features. The DDE and 2D techniques are used to configure all vector profiles. The dipeptide composition, as described by Bhasin and Raghava (2004) [<xref ref-type="bibr" rid="ref-34">34</xref>], has been employed to forecast various protein sequence functions, as demonstrated by Dhanda et al. (2013) [<xref ref-type="bibr" rid="ref-35">35</xref>].</p>
<p>Hence, the Dipeptide Deviation from Expected Mean (DDE) method, developed by Saravanan and Gautham (2015) [<xref ref-type="bibr" rid="ref-36">36</xref>], Quantifies deviations in dipeptide frequencies within protein sequences to different enzymes from non-enzymes. The DDE function vector is derived using three parameters: (1) Observed dipeptide composition (<italic>D</italic><sub><italic>c</italic></sub>), (2) theoretical mean (<inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>), and (3) theoretical variance (<inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>).
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mi>N</mml:mi></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>Here, <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represents the observed frequency of a dipeptide <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mi>i</mml:mi></mml:math></inline-formula> in the sequence, where <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the number of occurrences of the dipeptide <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mi>i</mml:mi></mml:math></inline-formula>, and <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mi>L</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> is the total number of dipeptides in a sequence of length <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mi>L</mml:mi></mml:math></inline-formula>.
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo>&#x00D7;</mml:mo><mml:mfrac><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>The theoretical mean (<inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>) estimates the expected frequency of a dipeptide <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mi>i</mml:mi></mml:math></inline-formula>, assuming independence between amino acids. <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> denote the counts of the first and second amino acid in dipeptide <italic>i</italic> across the entire dataset, while <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>61</mml:mn></mml:math></inline-formula> represents the total codons (excluding three stop codons).
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mi>N</mml:mi></mml:mfrac></mml:math></disp-formula></p>
<p>The theoretical variance (<inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>) quantifies the expected variability of a dipeptide&#x2019;s <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:mi>i</mml:mi></mml:math></inline-formula> frequency under the independence assumption, where <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mi>N</mml:mi></mml:math></inline-formula> is the total dipeptides in the sequence.
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mi>D</mml:mi><mml:mi>D</mml:mi><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msqrt><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:msqrt></mml:mfrac></mml:math></disp-formula></p>
<p>The <italic>DDE</italic> score standardizes the deviation of observed dipeptide frequencies from their theoretical expectations, with higher absolute values indicating stronger deviations. A 400-dimensional feature vector is generated by computing <italic>DDE</italic><sub>(<italic>i</italic>)</sub> for all 400 possible dipeptides.
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>D</mml:mi><mml:mi>D</mml:mi><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mi>D</mml:mi><mml:mi>D</mml:mi><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:msub><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:msub><mml:mo>.</mml:mo><mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:msub><mml:mo>&#x2026;</mml:mo><mml:mi>D</mml:mi><mml:mi>D</mml:mi><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>n</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>w</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mn>400</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Feature Extraction Using Enhanced Amino Acid Composition (EAAC)</title>
<p>The approach proposed according to Chen et al. [<xref ref-type="bibr" rid="ref-37">37</xref>] involves the extraction of sequential protein information, which is subsequently used to derive amino-acid frequency information.
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>H</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>n</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mi>D</mml:mi><mml:mo>,</mml:mo><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo><mml:mi>Y</mml:mi><mml:mo>}</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mi>W</mml:mi><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>W</mml:mi><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo><mml:mi>W</mml:mi><mml:mi>L</mml:mi><mml:mo>}</mml:mo></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>This method calculated the frequency of amino acid m within sliding windows of the sequence. Denotes the count of amino acid m in the <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mi>n</mml:mi><mml:mrow><mml:mtext>-</mml:mtext></mml:mrow><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:math></inline-formula> window, and <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mi>H</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is the window length. For example, a window size of 5 residues is slid across the sequence to capture local compositional biases.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Enhanced Grouped Amino Acid Composition (EGAAC)</title>
<p>The computation of EGAAC is derived from the categorization of amino acids. This study classifies amino acids into five distinct categories based on their physicochemical characteristics [<xref ref-type="bibr" rid="ref-38">38</xref>]. This method uses protein sequences to create numerical feature vectors, which are derived from their respective properties. Amino acid are grouped into five categories based on psychochemical properties: aliphatic (such as G, A, V, L, M, I), aromatic (F, Y, W), positively charged (K, R, H), negatively charged (D, E), and neutral (S, T, C, P, N, Q). While here (<inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:mi>H</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>) is the count of amino acids in group g within the <italic>n</italic>-<italic>th</italic> window, and <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:mi>H</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is the window length, for example, 5 residues. This feature extraction approach is highly effective in bioinformatics research fields, particularly in the prediction of druggable proteins. The calculation of EGAAC is determined by the following equation:
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mi>E</mml:mi><mml:mi>G</mml:mi><mml:mi>A</mml:mi><mml:mi>A</mml:mi><mml:mi>C</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>g</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>H</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>g</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>n</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>g</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mi>g</mml:mi><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>g</mml:mi><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>g</mml:mi><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow><mml:mi>g</mml:mi><mml:mn>4</mml:mn><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo><mml:mi>g</mml:mi><mml:mn>5</mml:mn><mml:mo>}</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>n</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mi>W</mml:mi><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>W</mml:mi><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo><mml:mi>W</mml:mi><mml:mi>L</mml:mi><mml:mo>}</mml:mo></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:math></disp-formula></p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Proposed Method Step by Step</title>
<p>Consequently, it is crucial to translate knowledge about protein sequences into numerical values that ensemble learning algorithms can comprehend and use [<xref ref-type="bibr" rid="ref-39">39</xref>&#x2013;<xref ref-type="bibr" rid="ref-41">41</xref>]. This study utilized three methods for extracting characteristics from protein sequences. The three methods are.</p>
<p>The DDE, EGAAC, and EAAC. An approach to machine learning known as ensemble learning involves training many models. Commonly referred to as &#x201C;weak learners,&#x201D; they address the same problems and then combine them to achieve improved outcomes. By combining weak models, we can achieve a more precise model. The concept of Ensemble Learning encompasses three distinct models, including Bagging, Boosting, and Stacking. The fundamental framework comprises multiple machine learning models. The methodology known as CEL-IDPs <xref ref-type="fig" rid="fig-1">Fig. 1</xref> comprises five fundamental steps: (i) The researchers gathered datasets and benchmarking data, followed by preprocessing and eliminating redundant similarity, duplication deletion. (ii) Features were extracted to cover various characteristics of sequencing data. (iii) A feature representation learning methodology was utilized, and analysis was conducted using the t-distributed Stochastic Neighbor Embedding (t-SNE) algorithm. (iv) A predictor was constructed using a three-step distinct model: Bagging, Boosting, and Stacking. (v) The final classification model was constructed, as depicted in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>. The following sections have presented a comprehensive elucidation of each of these noteworthy phases.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Workflow of the CEL-IDPs framework integrating feature extraction and ensemble learning for druggable protein prediction</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-1.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig-1">Fig. 1</xref> illustrates the proposed methodology&#x2019;s comprehensive flow chart that depicts the organized process of our proposed methodology for identifying druggable proteins. In Section A, we compiled and improved a dataset of both druggable and non-druggable proteins, assuring data quality through redundancy. Section B emphasizes feature analysis, wherein we do SHAP analysis to elucidate feature significance, followed by t-SNE for data visualization. Furthermore, we obtained essential protein sequence descriptors, namely DDE, EAAC, and CGAAC. Section C delineates the predictive modeling phase in which various ensemble learning techniques-bagging, boosting, and stacking were employed to improve model generalizability and classification efficacy. The initial phase was the construction of a database containing both druggable and non-druggable proteins. Subsequently, the calculation of three families of protein composition descriptors was performed using Python. These descriptors included the 20 amino acid composition (AC), the Dipeptide deviation from expected mean (DDE), features length was 400, the Enhanced amino acid composition (EAAC) method, features length was 920, and the enhanced grouped amino acid composition (EGAAC) method, features length was 230. In the subsequent phase, Jupyter notebooks were utilized to implement Python/Scikit-learn-based machine learning classifiers [<xref ref-type="bibr" rid="ref-42">42</xref>]. These classifiers were constructed by integrating thirteen distinct types, which were derived from the combination of three descriptor families (DDE, EAAC, EGAAC). The classifiers utilized in this study encompassed a range of machine learning algorithms. First step, we conducted all analyses using the Bagging method first (Bagging Classifier) [<xref ref-type="bibr" rid="ref-43">43</xref>], second, the random forest (Random Forest Classifier) [<xref ref-type="bibr" rid="ref-44">44</xref>], and third, the decision tree (Decision Tree Classifier) [<xref ref-type="bibr" rid="ref-45">45</xref>]. In the second step, we used quantitative techniques to analyze the Boosting method, specifically, first AdaBoost [<xref ref-type="bibr" rid="ref-45">45</xref>], GB Boost [<xref ref-type="bibr" rid="ref-46">46</xref>], XGBM [<xref ref-type="bibr" rid="ref-47">47</xref>], LGBM, and CatBoost classifiers are used for the prediction of DPs sites. In the third step, we used the stacking method to combine quantitative techniques for analysis, which was based on first LR, SVC, NB [<xref ref-type="bibr" rid="ref-48">48</xref>], KNN [<xref ref-type="bibr" rid="ref-49">49</xref>], and DT [<xref ref-type="bibr" rid="ref-50">50</xref>] classifiers for the prediction of DPs sites. XGBoost (XGB) is an alternative ensemble method that utilizes sequential weak trees to rectify classification errors [<xref ref-type="bibr" rid="ref-47">47</xref>]. Using gradient boosting to classify is a well-established bootstrapping technique that relies on utilizing a succession of consecutive weak classifiers. A meta-estimator called the Ada-Boost classifier (AdaBoost) [<xref ref-type="bibr" rid="ref-45">45</xref>] starts the fitting process using a classifier that was created from the original dataset. Subsequently, it incorporates many iterations pertaining to the initial classifier, each with modified weights to account for cases that were improperly categorized. The Bagging classifier, also known as Bagging, shares similarities with AdaBoost. In Bagging, supplementary classifiers are produced from sub-sets of the initial dataset. The prediction model for machine learning was developed with distinct protein datasets. The positive set comprised 1219 druggable proteins, identified through their inclusion in the DrugBank database (<ext-link ext-link-type="uri" xlink:href="https://www.drugbank.ca">www.drugbank.ca</ext-link>) and their correlation with FDA-approved pharmaceuticals. Conversely, the negative protein collection consisted of 1299 proteins that are not amenable to pharmacological targeting, as reported in a previous study.</p>

<p>The final machine learning prediction model was used to analyze three sets of cancer-associated protein lists. A total of 230 proteins were identified as essential in the BC dataset [<xref ref-type="bibr" rid="ref-51">51</xref>]. Additionally, 2353 proteins known to drive cancer were obtained from the Network of Cancer Genes [<xref ref-type="bibr" rid="ref-52">52</xref>], while 1365 proteins were classified as RNA-binding proteins (RBPs).</p>
<sec id="s2_5_1">
<label>2.5.1</label>
<title>Bagging Method</title>
<p>In this analysis, bagging is a popular ensemble learning technique that combines the predictive outcomes of various base classifiers to provide a robust final classification result. The integration technique involves acquiring training subsets by selecting from the initial dataset, with each subset used to train a distinct model. The classification outcomes of samples are derived by a voting approach. As shown in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>The bagging model framework combines three classifiers</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-2.tif"/>
</fig>
<p><list list-type="order">
<list-item>
<p>Bagging</p></list-item>
<list-item>
<p>Random Forest algorithm</p></list-item>
<list-item>
<p>Extra Trees</p></list-item>
</list></p>
<p>The Decision Tree model was chosen as the base model due to its consistent and outstanding performance in prior research studies [<xref ref-type="bibr" rid="ref-53">53</xref>&#x2013;<xref ref-type="bibr" rid="ref-55">55</xref>].</p>
<p>Bagging is a method that employs random sampling to divide the training data related to every base learner into subsets for training. The fundamental learners are aggregated through majority voting to form a robust classifier. The bagging method employed is detailed in Algorithm 1. The most prevalent example of bagging implementations is random forests.</p>
<fig id="fig-15">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-15.tif"/>
</fig>
</sec>
<sec id="s2_5_2">
<label>2.5.2</label>
<title>Boosting Method</title>
<p>Boosting is a machine learning technique that combines adaptive sequential learning with a single base model. The optimal outcomes are obtained by combining the outcomes of each basis model with the outcomes of the preceding base model, as seen in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>The boosting model framework combines five classifiers</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-3.tif"/>
</fig>
<p>The AdaBoost model, also known as Adaptive Boosting, was chosen to improve ensemble learning.
<list list-type="order">
<list-item>
<p>The acronym GBM refers to Gradient Boosting Machines.</p></list-item>
<list-item>
<p>Extra Gradient Boosting Machine (XGBM)</p></list-item>
<list-item>
<p>The Light Gradient Boosting Machine (LGBM) algorithms.</p></list-item>
<list-item>
<p>CatBoost is the fifth choice.</p></list-item>
</list></p>
<p>The Decision Tree model was chosen as the base model because it consistently produced excellent results in earlier research [<xref ref-type="bibr" rid="ref-53">53</xref>&#x2013;<xref ref-type="bibr" rid="ref-55">55</xref>].</p>
<p>Using decision trees as base learners with a single split, AdaBoost is the first and most successful boosting method. The best way to use AdaBoost, which is used for binary classification tasks, is with AdaBoost M1. The Algorithm 2 explains the boosting method.</p>
<fig id="fig-16">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-16.tif"/>
</fig>
</sec>
<sec id="s2_5_3">
<label>2.5.3</label>
<title>Stacking Method</title>
<p>Stacking is a method in Ensemble Learning wherein many base models are trained independently and simultaneously. The basic models are subsequently integrated through a meta-learning method to obtain the ultimate output, which is derived from the amalgamation of their respective findings. The design of the stacking model consists of many base models, termed level-0 models, in conjunction with a meta model that integrates the predictions of the base models, identified as a level-1 model, as illustrated in <xref ref-type="fig" rid="fig-4">Fig. 4</xref>. The stacking ensemble learning experiment will utilize five base models, specifically:</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>The stacking model architecture includes five classifiers</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-4.tif"/>
</fig>
<p><list list-type="order">
<list-item>
<p>Logistic Regression</p></list-item>
<list-item>
<p>Support Vector Machine (SVC)</p></list-item>
<list-item>
<p>Gaussian Naive Bayes</p></list-item>
<list-item>
<p>K-Nearest Neighbor algorithm</p></list-item>
<list-item>
<p>Decision Trees</p></list-item>
</list></p>
<p>Logistic regression has been selected as the meta-model, while the stacking procedure is described in Algorithm 3.</p>
<fig id="fig-17">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-17.tif"/>
</fig>
<p>The ML classifiers were constructed using a 5-fold cross-validation (CV) technique. A pipeline was established for every fold in the experiment. The test set was scaled to correspond with the same scale after the training set was scaled utilizing the Standard Scaler. The AUROC scores for these three ensemble learning methods in all splits were calculated using a cross-validation score. The mean values and standard deviation (SD) of the area under the receiver operating characteristic curve (AUROC) were reported for each machine learning classifier on the test subset. Following the utilization of a machine learning model to screen three protein sets associated with cancer, the proteins deemed suitable for drug targeting were subjected to analysis using g: Profiler (<ext-link ext-link-type="uri" xlink:href="https://biit.cs.ut.ee/gprofiler/">https://biit.cs.ut.ee/gprofiler/</ext-link>, accessed on 01 January 2025). This research aimed to deliver significant annotations (with a false discovery rate under 0.001) pertaining to gene ontology (GO) concepts, pathways, and disease phenotypes [<xref ref-type="bibr" rid="ref-39">39</xref>]. Circos plots were generated to visually represent the current status of clinical trials associated with the most potential druggable proteins identified by the Open Targets Platform (<ext-link ext-link-type="uri" xlink:href="https://www.targetvalidation.org">https://www.targetvalidation.org</ext-link>). The platform in question is a thorough data integration system that makes it easier to obtain and visualize potential targets for cancer treatment [<xref ref-type="bibr" rid="ref-40">40</xref>].</p>
</sec>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Model Performance Evaluation</title>
<p>Each model undergoes assessment through the metrics of Accuracy, Precision, and Recall. The values derived from the confusion matrix, which comprises true positive (TP), true negative (TN), false positive (FP), and false negative (FN) values, are used to compute these three measures. The computation of accuracy entails dividing the aggregate of true positives and true negatives by the overall data volume. It provides a comprehensive analysis of the outcomes achieved using the recommended methodology. The proposed technique is assessed in comparison with current methods to demonstrate its superior performance. The PR curves and the AUPRC are more informative regarding performance in class-imbalanced cases than the ROC curves in isolation. To add the PR curves and AUPRC scores to our performance evaluation. We added graphical PR-curves for the proposed and baseline methods in the revised manuscript, along with a paragraph underlining their importance in the presence of unbalanced data (Please see revised <xref ref-type="sec" rid="s4_10">Section 4.10</xref>). These findings further support the robustness of our method in recognizing druggable proteins, especially when it comes to imbalanced classes.
<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:mo movablelimits="true" form="prefix">Pr</mml:mo><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></disp-formula>
<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mrow><mml:mtext>Re</mml:mtext></mml:mrow><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></disp-formula>
<disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mtext>-</mml:mtext></mml:mrow><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mfrac><mml:mrow><mml:mo movablelimits="true" form="prefix">Pr</mml:mo><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mo movablelimits="true" form="prefix">Pr</mml:mo><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
</sec>
<sec id="s4">
<label>4</label>
<title>Results and Discussion</title>
<p>The project aims to propose novel classification models for the prediction of newly druggable proteins. This will be achieved by utilizing three families of protein composition descriptors, namely DDE, EAAC, and EGAAC, which are derived using the ensemble learning methods. Jupyter notebooks, utilizing Python and the sklearn library, were employed to construct classifiers employing a total of thirteen distinct machine learning classifiers and five distinct feature selection approaches. Various parameters were employed in this process, as depicted in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>. The categorization performance was quantified using AUROC in the scripts. We conducted experiments using models that had varying numbers of features, specifically 400, 920, and 230, as well as models with a combination of different feature quantities. The AUROC values provided in this study are the mean values obtained after a five-fold cross-validation process.</p>

<sec id="s4_1">
<label>4.1</label>
<title>Ensemble Learning Methods for Accuracy Comparison Results</title>
<p>In <xref ref-type="table" rid="table-2">Table 2</xref>, the results of our study indicate a requirement for improved outcomes in order to demonstrate that the proposed methodologies of bagging, boosting, and stacking improve the performance of these three feature extraction methods. DDE, EAAC, and EGAAC approaches. Results were considered significant. DDE features were used with bagging (RF obtained 66.87%), boosting (XGBM obtained 66.66%), and stacking (Stacking obtained a score of 68.33%). After that, EAAC features used bagging (Bagging obtained 66.66%), boosting (CatBoost obtained 56.66%), and stacking (Stacking obtained a score of 66.66%). Third EGAAC features were considered significant and outstanding, with bagging (RF obtained 71.66%), boosting (XGBM obtained 71.66%), and stacking (Stacking obtained a score of 68.33%). The highest level of performance is achieved by employing the EGAAC feature extraction method with bagging, boosting, and stacking methods.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Accuracy comparison using the ensemble learning method results</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Features</th>
<th>Method</th>
<th>Model</th>
<th>Accuracy</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3"><bold>DDE</bold></td>
<td>Bagging</td>
<td>RF</td>
<td>66.87</td>
</tr>
<tr>
   
<td>Boosting</td>
<td>XGBM</td>
<td>66.66</td>
</tr>
<tr>

<td>Stacking</td>
<td>Stacking</td>
<td>68.33</td>
</tr>
<tr>
<td rowspan="3"><bold>EAAC</bold></td>
<td>Bagging</td>
<td>Bagging</td>
<td>66.66</td>
</tr>
<tr>

<td>Boosting</td>
<td>CatBoost</td>
<td>56.66</td>
</tr>
<tr>

<td>Stacking</td>
<td>Stacking</td>
<td>66.66</td>
</tr>
<tr>
<td rowspan="3"><bold>EGAAC</bold></td>
<td>Bagging</td>
<td>Bagging</td>
<td>71.66</td>
</tr>
<tr>

<td>Boosting</td>
<td>XGBM</td>
<td>71.66</td>
</tr>
<tr>

<td>Stacking</td>
<td>Stacking</td>
<td>68.33</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Ensemble Learning Methods for DDE Features Performance Comparison Results</title>
<p>Results were considered significant. DDE features were used with bagging (RF obtained 66.87%), boosting (XGBM obtained 66.66%), and stacking (Stacking obtained a score of 68.66%), as shown in <xref ref-type="table" rid="table-3">Table 3</xref>. The initial data set, referred to as the independent dataset, was created using the DDE model and the bagging, boosting, and stacking strategies. <xref ref-type="table" rid="table-3">Table 3</xref> illustrates this. The stacking method exhibits the greatest accuracy among ensemble learning techniques, attaining an accuracy score of 68.33%, as illustrated in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>. Additionally, the stacking process produces precise outcomes.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>DDE features performance comparison results (Bagging, Boosting, and Stacking)</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Features</th>
<th>Method</th>
<th>Model</th>
<th>Accuracy</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="9"><bold>DDE</bold></td>
<td rowspan="3">Bagging</td>
<td>Bagging</td>
<td>66.66</td>
</tr>
<tr>


<td>RF</td>
<td><bold>66.87</bold></td>
</tr>
<tr>


<td>ET</td>
<td>66.66</td>
</tr>
<tr>

<td rowspan="5">Boosting</td>
<td>AdaBoost</td>
<td>58.33</td>
</tr>
<tr>


<td>GBM</td>
<td>63.33</td>
</tr>
<tr>


<td>XGBM</td>
<td><bold>66.66</bold></td>
</tr>
<tr>


<td>LGBM</td>
<td>65.66</td>
</tr>
<tr>


<td>CatBoost</td>
<td>65.82</td>
</tr>
<tr>

<td>Stacking</td>
<td>Stacking</td>
<td><bold>68.33</bold></td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>DDE features performance comparison results (Bagging, Boosting, and Stacking)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-5.tif"/>
</fig>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Ensemble Learning Methods for EAAC Features Performance Comparison Results</title>
<p>After that, EAAC features used bagging (Bagging obtained 66.66%), boosting (CatBoost obtained 56.66%), and stacking (Stacking obtained a score of 66.66%), as shown in <xref ref-type="table" rid="table-4">Table 4</xref>. This is feasible due to the significantly larger amount of data in the dataset compared with dataset 1. It makes it possible for the model to enhance its learning efficiency, as illustrated in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>. Furthermore, it is possible that the data employed has a distinct pattern. According to the experimental results on the dataset, the boosting methods AdaBoost, XGB, and Light Gradient Boosting, as well as the random forest Bagging model, attained an average accuracy of 56.66%. Boosting has greater accuracy than bagging and stacking, as evidenced by the prior dataset, with the stacking method achieving the highest accuracy at 66.66%.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>EAAC features performance comparison results (Bagging, Boosting, and Stacking)</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Features</th>
<th>method</th>
<th>Model</th>
<th>Accuracy</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="9"><bold>EAAC</bold></td>
<td>Bagging</td>
<td>Bagging</td>
<td>66.66</td>
</tr>
<tr>

<td>Bagging</td>
<td>RF</td>
<td>66.66</td>
</tr>
<tr>

<td>Bagging</td>
<td>ET</td>
<td>66.66</td>
</tr>
<tr>

<td>Boosting</td>
<td>AdaBoost</td>
<td>46.66</td>
</tr>
<tr>

<td>Boosting</td>
<td>GBM</td>
<td>51.66</td>
</tr>
<tr>

<td>Boosting</td>
<td>XGBM</td>
<td>55.66</td>
</tr>
<tr>

<td>Boosting</td>
<td>LGBM</td>
<td>56.25</td>
</tr>
<tr>

<td>Boosting</td>
<td>CatBoost</td>
<td>56.66</td>
</tr>
<tr>

<td>Stacking</td>
<td>Stacking</td>
<td>66.66</td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>EAAC features performance comparison results (Bagging, Boosting, and Stacking)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-6.tif"/>
</fig>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Ensemble Learning Methods for EGAAC Features Performance Comparison Results</title>
<p>Third EGAAC features were considered significant and outstanding results with bagging (RF obtained 71.66%), boosting (XGBM obtained 71.66%), and stacking (Stacking obtained a score of 68.33%), as shown in <xref ref-type="table" rid="table-5">Table 5</xref>. The highest level of performance is achieved by employing the EGAAC feature extraction method using bagging, boosting, and stacking techniques, as illustrated in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>. <xref ref-type="table" rid="table-5">Table 5</xref> displays the results of ensemble learning experiments performed on a separate dataset utilizing the EGAAC model. The accuracy acquired from the experimental findings of three ensemble learning approaches, specifically Bagging, Boosting, and Stacking, varied across three distinct datasets.</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>EGAAC features performance comparison results (Bagging, Boosting, and Stacking)</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Features</th>
<th>method</th>
<th>Model</th>
<th>Accuracy</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="9"><bold>EGAAC</bold></td>
<td>Bagging</td>
<td>Bagging</td>
<td>71.66</td>
</tr>
<tr>

<td>Bagging</td>
<td>RF</td>
<td>70.65</td>
</tr>
<tr>

<td>Bagging</td>
<td>ET</td>
<td>71.66</td>
</tr>
<tr>

<td>Boosting</td>
<td>AdaBoost</td>
<td>58.33</td>
</tr>
<tr>

<td>Boosting</td>
<td>GBM</td>
<td>66.33</td>
</tr>
<tr>

<td><bold>Boosting</bold></td>
<td><bold>XGBM</bold></td>
<td><bold>71.66</bold></td>
</tr>
<tr>

<td><bold>Boosting</bold></td>
<td><bold>LGBM</bold></td>
<td><bold>63.33</bold></td>
</tr>
<tr>

<td>Boosting</td>
<td>CatBoost</td>
<td>70.72</td>
</tr>
<tr>

<td>Stacking</td>
<td>Stacking</td>
<td>68.33</td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>EGAAC features performance comparison results (Bagging, Boosting, and Stacking)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-7.tif"/>
</fig>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Comprehensive Evaluation and Interpretability of Classifiers</title>
<sec id="s4_5_1">
<label>4.5.1</label>
<title>Classifier Performance Comparison</title>
<p>To evaluate the classification efficacy of various machine learning methods, as shown in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>, we analyzed the performance of many base classifiers and ensemble models employing common evaluation metrics: accuracy, precision, recall, F1-score, and ROC AUC. <xref ref-type="table" rid="table-6">Table 6</xref> summarizes the findings for nine classifiers: Random Forest, Gradient Boosting, Support Vector Machine (SVM), Logistic Regression, Decision Tree, Voting (both Hard and Soft), Bagging, and Stacking. The Stacking classifier exhibited the highest overall performance among all models, achieving an F1-score of 0.6683 and a ROC AUC of 0.7219. Subsequently, Soft Voting and Random Forest classifiers were thoroughly examined. These findings underscore the efficacy of ensemble learning methodologies in improving prediction accuracy and model resilience for the identification of druggable proteins.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Visual comparison of classifier performance using evaluation metrics</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-8.tif"/>
</fig><table-wrap id="table-6">
<label>Table 6</label>
<caption>
<title>Performance comparison of classifiers using accuracy, precision, recall, F1-Score, and ROC AUC metrics</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th></th>
<th>Model</th>
<th>Accuracy</th>
<th>Precision</th>
<th>Recall</th>
<th>F1 Score</th>
<th>ROC AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="9"><bold>Performance comparison</bold></td>
<td>Random forest</td>
<td>0.6607</td>
<td>0.6605</td>
<td>0.660714</td>
<td>0.660316</td>
<td>0.716275</td>
</tr>
<tr>

<td>Gradient boosting</td>
<td>0.6250</td>
<td>0.6249</td>
<td>0.625</td>
<td>0.624975</td>
<td>0.686326</td>
</tr>
<tr>

<td>SVM</td>
<td>0.6587</td>
<td>0.6593</td>
<td>0.65873</td>
<td>0.657253</td>
<td>0.712334</td>
</tr>
<tr>

<td>Decision tree</td>
<td>0.5912</td>
<td>0.5919</td>
<td>0.59127</td>
<td>0.591366</td>
<td>0.591488</td>
</tr>
<tr>

<td>Logistic regression</td>
<td>0.6388</td>
<td>0.6387</td>
<td>0.638889</td>
<td>0.638158</td>
<td>0.671761</td>
</tr>
<tr>

<td>Voting (Hard)</td>
<td>0.6468</td>
<td>0.6466</td>
<td>0.646825</td>
<td>0.646357</td>
<td>0.671761</td>
</tr>
<tr>

<td>Voting (Soft)</td>
<td>0.6567</td>
<td>0.6565</td>
<td>0.656746</td>
<td>0.656441</td>
<td>0.718324</td>
</tr>
<tr>

<td>Stacking</td>
<td>0.6686</td>
<td>0.6685</td>
<td>0.668651</td>
<td>0.668262</td>
<td>0.721887</td>
</tr>
<tr>

<td>Bagging</td>
<td>0.6190</td>
<td>0.6195</td>
<td>0.619048</td>
<td>0.616473</td>
<td>0.63694</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_5_2">
<label>4.5.2</label>
<title>Confusion Matrix Analysis</title>
<p>We visualized the confusion matrices for selected classifiers to conduct a more in-depth analysis of model performance. These matrices depict the quantities of true positives, true negatives, false positives, and false negatives, offering a comprehensive assessment of each model&#x2019;s efficacy in differentiating between druggable and non-druggable proteins. <xref ref-type="fig" rid="fig-9">Fig. 9</xref> illustrates that the Stacking and Soft Voting classifiers yield the most equitable classification outcomes, exhibiting superior true positive and true negative rates relative to other models. These visual diagnostics validate the efficacy and generalizability of ensemble-based approaches in protein classification.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>Confusion matrix visualizations for selected classifiers</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-9a.tif"/>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-9b.tif"/>
</fig>
</sec>
<sec id="s4_5_3">
<label>4.5.3</label>
<title>Feature Importance from Random Forest</title>
<p>We evaluated feature relevance using the Random Forest model to determine which input features most significantly contributed to the classification process. Importance scores were derived using the Gini impurity criterion. As shown in <xref ref-type="fig" rid="fig-10">Fig. 10</xref>, many of the top-ranked features originated from the EGAAC encoding method, highlighting the biochemical significance of amino acid classifications based on physicochemical groupings. These features substantially improved the model&#x2019;s ability to distinguish between druggable and non-druggable proteins, offering critical insights into model interpretability and offering critical insights into model interpretability.</p>
<fig id="fig-10">
<label>Figure 10</label>
<caption>
<title>Feature importance plot from the random forest model</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-10.tif"/>
</fig>
<p>The magnitude levels of the attributes are plotted against the axis for the purposes of the discussion. The scores are computed based on the Gini impurity criterion and indicate the contribution of a feature in the classification process. A higher number indicates the characteristic is more important. <italic>Y</italic>-axis (Features): Each row represents one of the features from the dataset. The longer the line length, the more important this feature is in determining the category.</p>
<p>RHS: The features in contrast (RHS) have higher importance ratings, and they have a greater impact on the decision process of the machine. These features likely closely relate to the output class (eg, distinguishing between druggable and non-druggable proteins in your case). Left side features that are less significant and contribute little to the classification. Rightward extending features (which correspond to high importance) probably originated from the EGAAC encoding procedure, as discussed in the text. The physicochemical properties of the amino acid residues seem to heavily influence the prediction of druggability. Leftward features with shorter bars may be less important or may be redundant information for making precise predictions.</p>
</sec>
</sec>
<sec id="s4_6">
<label>4.6</label>
<title>Comparative Analysis with Existing Druggable Proteins Prediction</title>
<p>Prioritizing previously identified genes based on the Pharmacogenomics Knowledgebase (PharmGKB), the interpreter for cancer genomes, and the Consensus Strategy of the Pan-Cancer Atlas project, we conducted comprehensive analyses of genetic changes, signaling pathways, protein interactions, networks, protein expression, dependency maps, and enrichment maps in a prior study. These investigations enabled us to identify critical proteins linked to the etiology of breast cancer. Predicting the draggability of breast cancer (BC) proteins using machine learning (ML) techniques has the potential to provide important information about potential biomarkers, potential treatment targets, and upcoming clinical trials. This strategy could improve precision medicine and global cancer pharmacogenomics while reducing ethnic bias [<xref ref-type="bibr" rid="ref-56">56</xref>,<xref ref-type="bibr" rid="ref-57">57</xref>].</p>
</sec>
<sec id="s4_7">
<label>4.7</label>
<title>SHAP-Based Feature Analysis for Druggable Protein Identification</title>
<p>To enhance the interpretability of our predictive model, we utilized SHAP (SHapley Additive exPlanations) to analyze the impact of specific characteristics in distinguishing druggable and non-druggable proteins. Unlike traditional feature selection methods, through the quantification of each feature&#x2019;s influence on the model&#x2019;s decision-making mechanism, SHAP offers a simple way to gauge the value of features [<xref ref-type="bibr" rid="ref-58">58</xref>].</p>
<p><xref ref-type="fig" rid="fig-11">Fig. 11</xref> illustrates the SHAP-based feature analysis applied to three different feature encoding methods: Dipeptide Deviation Encoding (DDE), Enhanced Amino Acid Composition (EAAC), and Extended Grouped Amino Acid Composition (EGAAC). The SHAP plots reveal the most influential features contributing to the classification process. Features with high SHAP values strongly influence the model&#x2019;s decision in favor of druggable proteins (positive impact) or non-druggable proteins (negative impact). These findings offer deeper insights into the biochemical properties that play a crucial role in druggability prediction, thereby strengthening the biological relevance of our model.</p>
<fig id="fig-11">
<label>Figure 11</label>
<caption>
<title>SHAP-based feature (Bagging, Boosting, and Stacking)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-11.tif"/>
</fig>
<p>The figure demonstrates SHAP (SHapley Additive exPlanations)-based feature analysis applied to different datasets (DDE, EAAC, and EGAAC) using three different machine learning techniques: Bagging, Boosting, and Stacking. While SHAP values explain the importance of features for the model&#x2019;s prediction, the figure does not make the relevant biology behind these traits as they relate to drug targets for identification, as you observed. Guided visualization and interpretation of SHAP-Inspired results: To improve the interpretability and utility of the SHAP-based research image shown in this study, it is important to annotate these top-ranked features (by the SHAP because these features are considered top-ranked features) with known biology and biological relationships. Conduct checks on the literature to ensure that these top-ranking features (e.g., &#x201C;DDE115,&#x201D; &#x201C;EAAC616,&#x201D; &#x201C;EGAAC146,&#x201D; etc.) actually correspond to biological functions or diseases that are relevant to the new drug target prediction. Explore whether these characteristics are linked to known biomarkers, genes, proteins, or pathways related to diseases such as cancer, diabetes, or neurodegenerative diseases. A functional relationship can be established if the features in the datasets are related to gene expression or protein activities; otherwise, use Gene Ontology (GO) analysis or pathway databases (e.g., KEGG, Reactome, or STRING). For example, one might question whether the genes linked to the affected traits are involved in specific biological processes like cell growth, death, or immunological response. Use bioinformatics techniques to determine if these features are similar to known functional domains, molecular functions, or biological processes. Often, highly maintainable features in biological datasets represent core regulatory proteins, enzymes, and receptors that are involved in drug-target interactions. Cross-reference resulting core traits with biological databases (e.g., UniProt, Ensembl, and/or GeneCards) to identify their implication in established pharmacological targets. This might provide strong support that these core characteristics may serve as potential targets for therapy. Once features are associated with biological pathways or molecular mechanisms, experimental validation such as CRISPR-based knockdown experiments or small molecules screen can validate the potentiality of the features as therapeutic targets. This approach, by linking SHAP-based features to biological literature, can contribute to a better understanding of their impact on drug discovery and possibly help identify new drug targets. It would also improve the interpretability of the SHAP study and bridge computational models with biological insights.</p>
</sec>
<sec id="s4_8">
<label>4.8</label>
<title>t-SNE Visualization of Feature Space</title>
<p>To further investigate the separability of druggable and non-druggable proteins in the learned feature space, the nonlinear dimensionality reduction method t-Distributed Stochastic Neighbor Embedding (t-SNE) was utilized [<xref ref-type="bibr" rid="ref-59">59</xref>]. <xref ref-type="fig" rid="fig-12">Fig. 12</xref> presents the t-SNE projections for the DDE, EAAC, and EGAAC feature encodings, providing a 2D visualization of the distribution of protein sequences. Each dot in the scatter plots represents a protein, with druggable proteins marked in blue and non-druggable proteins in red. The clustering patterns observed in the t-SNE plots suggest that different feature representations have varying degrees of discriminatory power. While some encodings exhibit better separation between druggable and non-druggable proteins, others display overlapping distributions, indicating that feature selection and fusion strategies may further improve classification performance.</p>
<fig id="fig-12">
<label>Figure 12</label>
<caption>
<title>t-SNE Analysis (Bagging, Boosting, and Stacking)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-12.tif"/>
</fig>
</sec>
<sec id="s4_9">
<label>4.9</label>
<title>PCA Feature Analysis for Druggable Protein Identification</title>
<p>To further analyze the underlying structure of the collected features and enhance the t-SNE and SHAP-based interpretations, Principal Component Analysis (PCA) was performed on all three feature types: DDE, EAAC, and EGAAC. PCA is a well-established linear dimensionality reduction method that converts high-dimensional data into a lower-dimensional space, preserving the directions of highest variance using orthogonal principle components [<xref ref-type="bibr" rid="ref-60">60</xref>].</p>
<p><xref ref-type="fig" rid="fig-13">Fig. 13</xref> illustrates that the initial two principal components (PC1 and PC2) were employed to depict the class distribution across each feature set. Indicating that this descriptor captures discriminative patterns in the feature space, the PCA projection for DDE features showed a comparatively clearer separation between druggable and non-druggable proteins. Despite some reported class overlap in EAAC and EGAAC descriptors, significant grouping trends remained apparent. This indicates that even with a linear projection, the derived features retain class-relevant signals.</p>
<fig id="fig-13">
<label>Figure 13</label>
<caption>
<title>PCA analysis (Bagging, Boosting, and Stacking)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-13.tif"/>
</fig>
<p>Although PCA may not properly represent intricate nonlinear manifolds like t-SNE, its linear projections provide an interpretable and complementary view of the spatial distribution of samples in the feature space. The congruence of PCA visualizations with t-SNE mappings and SHAP-derived feature importance substantiates the validity and consistency of our engineered feature sets across various interpretive dimensions, emphasizing the robustness of the proposed feature design strategy and its appropriateness for druggable protein prediction, even when assessed through different dimensionality reduction methodologies.</p>
</sec>
<sec id="s4_10">
<label>4.10</label>
<title>AUC-PR Curve Performance for Druggable Protein Identification</title>
<p>This figure shows how various models (Random Forest, Gradient Boosting, Support Vector Machine, Decision Tree, Logistic Regression, Voting (Soft), Stacking, Bagging) performed on the Precision-Recall (PR) curve. Each of these models has an AUC-PR score. Here is an overview of the AUC-PR comparison on all models. The best AUC-PR is achieved by RF, Voting (Soft), and Stacking (0.71 for all these models) in this competition. Bagging scores the worst with an AUC-PR of 0.61. Area Under the Precision-Recall Curve (AUC-PR) The AUC-PR is a performance metric used to evaluate how well a model can distinguish between positive and negative classes, particularly for imbalanced datasets, such as that employed in Druggable Protein Identification (refer to <xref ref-type="fig" rid="fig-14">Fig. 14</xref>). The precision-recall curve (PR curve) allows us to understand to what degree the models are identifying the &#x201C;druggable&#x201D; proteins (positive class) vs. controlling the trade-off between minimizing the false positives (which are the proteins that are incorrectly labeled as druggable) and maximizing the precision (which is the correct recognition of druggable proteins). AUC-PR Curve Summary: Random Forest (AUC-PR &#x003D; 0.71): It does a great job at retrieving the druggable proteins; the performance is high at all recall values. This indicates that RF does a good job of identifying druggable proteins while avoiding increasing the number of false positives. Gradient Boosting (AUC-PR &#x003D; 0.69): The performance of this model is a bit lower than the Random Forest model, but it is still able to distinguish between druggable and non-druggable proteins quite well. Its performance isn&#x2019;t as good as Random Forest at first glance, but it&#x2019;s still good. SVM (AUC-PR &#x003D; 0.63): Doesn&#x2019;t crush Random Forest or Gradient Boosting, for sure. The plot visually demonstrates that it struggles in reducing the false positive rate, which translates into lower precision across different levels of recall. Decision Tree (A-UC-PR &#x003D; 0.68): Better than SVM but not as good as Random Forest or Gradient Boosting. The curve indicates that the Decision Tree has difficulty in finding druggable proteins, especially when a large recall is achieved. Logistic Regression (AUC-PR &#x003D; 0.65): It doesn&#x2019;t work as well as the tree-based models (Random Forest, Gradient Boosting, Decision Tree). It has a smaller AUC-PR, indicating a less effective balance of precision and recall. Voting (Soft) (AUC-PR &#x003D; 0.71): The Voting model (Soft) performs similarly to Random Forest. By applying a soft voting ensemble to average the predictions of multiple models, we can expect good performance while recognizing druggable proteins. Stacking (AUC-PR &#x003D; 0.71): Stacking, similar to Voting (Soft), takes the best of a few models and has a good AUC-PR.</p>
<fig id="fig-14">
<label>Figure 14</label>
<caption>
<title>AUC-PR curve performance</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_67412-fig-14.tif"/>
</fig>
<p>Under this setting, the best performer in these figures, Random Forest and Voting (Soft), demonstrates similar results. Bagging (AUC-PR &#x003D; 0.61): This model shows the lowest AUC-PR value, indicating lower AUC-PR value, which indicates less competitiveness and more false positives. This indicates that Drn can identify druggable proteins with more difficulty than the other models in the study. The models, such as Random Forest, Voting (Soft), and Stacking, perform the best in terms of predicting druggable proteins, with the AUC-PR score of 0.71. They strike the best balance between precision and recall, crucial for finding proteins that drugs can target. Bagging, in contrast, does horribly, with AUC-PR well lower, which implies it may not be as good at discovering druggable proteins.</p>
</sec>
<sec id="s4_11">
<label>4.11</label>
<title>AUC-PR Curve or MCC (Matthews Correlation Coefficient) Metrics for Druggable Protein Identification</title>
<p>Description of <xref ref-type="table" rid="table-7">Table 7</xref>: Return on Capital, AUC-PR, and MCC (Matthews Correlation Coefficient) metric scores. This table displays the evaluation measures AUC-PR (Area Under the Precision-Recall Curve) and MCC (Matthews Correlation Coefficient) for various model performances. These metrics are frequently employed in machine learning to evaluate the efficacy of categorization algorithms. This is an analysis of the columns and values in the table:</p>
<table-wrap id="table-7">
<label>Table 7</label>
<caption>
<title>ROC AUC-PR and MCC (Matthews correlation coefficient) metrics score</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th></th>
<th>AUC-PR</th>
<th>MCC</th>
</tr>
</thead>
<tbody>
<tr>
<td><bold>0</bold></td>
<td>0.711204</td>
<td>0.320107</td>
</tr>
<tr>
<td><bold>1</bold></td>
<td>0.691638</td>
<td>0.249151</td>
</tr>
<tr>
<td><bold>2</bold></td>
<td>0.692999</td>
<td>0.316226</td>
</tr>
<tr>
<td><bold>3</bold></td>
<td>0.683804</td>
<td>0.18289</td>
</tr>
<tr>
<td><bold>4</bold></td>
<td>0.647176</td>
<td>0.276148</td>
</tr>
<tr>
<td><bold>5</bold></td>
<td>NaN</td>
<td>0.2922</td>
</tr>
<tr>
<td><bold>6</bold></td>
<td>0.714455</td>
<td>0.312219</td>
</tr>
<tr>
<td><bold>7</bold></td>
<td>0.714057</td>
<td>0.336033</td>
</tr>
<tr>
<td><bold>8</bold></td>
<td>0.608063</td>
<td>0.236226</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The initial column, denoted by numbers 0 to 8, serves as an index that presumably signifies several models or experiments for which the metrics are calculated. This statistic assesses the balance between precision and recall at various thresholds, however the area under the curve (AUC) provides a comprehensive evaluation of model performance across all potential classification levels. Elevated AUC-PR values signify superior performance, with values approaching 1 denoting an ideal model and values nearing 0 reflecting subpar performance. MCC quantifies the quality of binary classifications. It considers both true positives and true negatives, and is seen as more relevant than other accuracy metrics when addressing imbalanced datasets. MCC values span from &#x2212;1 to &#x002B;1, with &#x002B;1 denoting a flawless forecast, 0 signifying a random guess, and &#x2212;1 representing a wholly erroneous prediction. MCC: 0.320107 A moderate MCC score, signifying acceptable model performance, if not exemplary. Analogous to the AUC-PR score, the MCC value is marginally lower, indicating a relatively diminished model performance. The MCC of 0.316226 surpasses that of Index 1, signifying a marginally superior classification accuracy. This table provides a summary of model performance across various metrics (AUC-PR and MCC). It emphasizes areas where specific models exhibit superior or inferior performance, and also illustrates the effect of absent data (NaN values) on various models.</p>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusions</title>
<p>The present investigation introduces a novel predictive framework for identifying druggable proteins. This framework leverages amino acid composition descriptors derived from protein sequences as input for three distinct Ensemble Learning Methods for Identification of Druggable Proteins (CEL-IDPs). The Stacking method is the most optimal method while utilizing a limited set of 400 dipeptide deviation from expected mean (DDE) amino acid composition characteristics.</p>
<p>The study identified the top 10 essential proteins in BC OncoOmics that are predicted to be druggable. These proteins include CDK4, AP1S1, POLE, HMMR, RPL5, PALB2, TIMP1, RPL22, NFKB1, and TOP2A. Additionally, the top 10 cancer-driving proteins that are predicted to be druggable were identified as TLL2, FAM47C, SAGE1, HTR1E, MACC1, ZFR2, VMA21, DUSP9, CTNNA3, and ABRG1. Lastly, the study also identified the top 10 RNA-binding proteins that are predicted to be druggable, which include PLA2G1B, CPEB2, NOL6, LRRC47, CTTN, CORO1A, SCAF11, KCTD12, and TMPO. However, the majority of these interventions currently need clinical trials. Ultimately, this robust model offers predictions for a number of proteins that have the potential to be targeted by drugs. These proteins warrant extensive investigation in order to identify more effective therapeutic targets and thereby enhance the efficacy of clinical trials.</p>
<p>This study presents a framework, CEL-IDP, which integrates sequence-derived features with three ensemble learning strategies. It identifies top druggable protein candidates and applies SHAP and t-SNE for model interpretability and visualization. Among all tested combinations, the ensemble models using EGAAC features, especially Bagging and Boosting, consistently achieved the highest accuracy, highlighting the robustness of these methods in predicting druggable proteins.</p>
<p>Various feature extraction algorithms can be employed to turn sequential nominal character information into a numerical vector. The enhancement of classification performance can be achieved through the extraction of efficient features. The following algorithms are utilized to extract characteristics from protein sequences: The EAAC refers to the increased amino acid composition. The improved grouped amino acid composition (EGAAC) and dipeptide deviation from expected mean (DDE) are two important factors to consider.</p>
<p>Clinical validation is required: Even though druggable proteins are predicted in this study, these targets still have to be tested through clinical trials for the proof of efficacy or relevance as therapeutic targets. Feature limitation: The model relies only on a limited set of 400 DDE amino acid composition attributes, which may be insufficient to capture the complexity of protein-drug association and is further limited by the magnitude of predictions. The generalizability of findings is limited by the focus of the study on specific sets of data, e.g., BC OncoOmics and cancer-driving proteins, limiting the usability of the model in other biological contexts or conditions. The approach depends heavily on sequence-based features, which might not fully capture other structural and dynamic features of proteins that may influence druggability.</p>
<p>Necessary for Further Examination: The top choice models and ensemble strategies (Bagging, Boosting, Stacking) all achieve superior accuracy, but the model needs more testing and refinement to ensure applicability to a broader range of proteins and conditions. Interpretability Shortcomings: SHAP and t-SNE enhance model interpretability, but still, they might not provide a complete or clear understanding of the complex relations natured in connection with these complex interactions between proteins and drugs or medicinal chemicals.</p>
</sec>
</body>
<back>
<ack>
<p>This research was supported by the MSIT (Ministry of Science and ICT), Korea, under the ITRC (Information Technology Research Centre) support program (IITP-2024-RS-2024-00437191) supervised by the IITP (Institute for Information &#x0026; Communications Technology Planning &#x0026; Evaluation).</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>This research was supported by the MSIT (Ministry of Science and ICT), Korea, under the ITRC (Information Technology Research Centre) support program (IITP-2024-RS-2024-00437191) supervised by the IITP (Institute for Information &#x00026; Communications Technology Planning &#x00026; Evaluation).</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>Conceptualization, Mujeebu Rehman and Qinghua Liu; methodology, Mujeebu Rehman and Ali Ghulam; software, Tariq Ahmad; validation, Mujeebu Rehman, Tariq Ahmad and Jawad Khan; formal analysis, Dildar Hussain; investigation, Jawad Khan; resources, Qinghua Liu; data curation, Yeong Hyeon Gu; writing original draft preparation, Mujeebu Rehman and Qinghua Liu; writing review and editing, Tariq Ahmad, Jawad Khan and Dildar Hussain; visualization, Mujeebu Rehman; supervision, Qinghua Liu; project administration, Qinghua Liu and Jawad Khan; funding acquisition, Dildar Hussain. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>We used the benchmark online freely available datasets that are mentioned in <xref ref-type="sec" rid="s2">Section 2</xref>.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kellner</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Proteomics. Concepts and perspectives</article-title>. <source>Fresenius&#x2019; J Anal Chem</source>. <year>2000</year>;<volume>366</volume>(<issue>6</issue>):<fpage>517</fpage>&#x2013;<lpage>24</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s002160051547</pub-id>; <pub-id pub-id-type="pmid">11225764</pub-id></mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Barbarino</surname> <given-names>JM</given-names></string-name>, <string-name><surname>Whirl-Carrillo</surname> <given-names>M</given-names></string-name>, <string-name><surname>Altman</surname> <given-names>RB</given-names></string-name>, <string-name><surname>Klein</surname> <given-names>TE</given-names></string-name></person-group>. <article-title>PharmGKB: a worldwide resource for pharmacogenomic information</article-title>. <comment>[cited 2025 Aug 12]</comment>. Available from: <ext-link ext-link-type="uri" xlink:href="https://wires.onlinelibrary.wiley.com/doi/10.1002/wsbm.1417">https://wires.onlinelibrary.wiley.com/doi/10.1002/wsbm.1417</ext-link>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Whirl-Carrillo</surname> <given-names>M</given-names></string-name>, <string-name><surname>McDonagh</surname> <given-names>EM</given-names></string-name>, <string-name><surname>Hebert</surname> <given-names>JM</given-names></string-name>, <string-name><surname>Gong</surname> <given-names>L</given-names></string-name>, <string-name><surname>Sangkuhl</surname> <given-names>K</given-names></string-name>, <string-name><surname>Thorn</surname> <given-names>CF</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Pharmacogenomics knowledge for personalized medicine</article-title>. <source>Clin Pharmacol Ther</source>. <year>2012</year>;<volume>92</volume>(<issue>4</issue>):<fpage>414</fpage>&#x2013;<lpage>7</lpage>. doi:<pub-id pub-id-type="doi">10.1038/clpt.2012.96</pub-id>; <pub-id pub-id-type="pmid">22992668</pub-id></mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Vefghi</surname> <given-names>A</given-names></string-name>, <string-name><surname>Rahmati</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Akbari</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Drug-target interaction/affinity prediction: deep learning models and advances review</article-title>. <source>Comput Biol Med</source>. <year>2025</year>;<volume>196</volume>(<issue>Pt A</issue>):<fpage>110438</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.compbiomed.2025.110438</pub-id>; <pub-id pub-id-type="pmid">40609289</pub-id></mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ali</surname> <given-names>N</given-names></string-name>, <string-name><surname>Hanif</surname> <given-names>N</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>HA</given-names></string-name>, <string-name><surname>Waseem</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Saeed</surname> <given-names>A</given-names></string-name>, <string-name><surname>Zakir</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Deep learning and artificial intelligence for drug discovery, application, challenge, and future perspectives</article-title>. <source>Discov Appl Sci</source>. <year>2025</year>;<volume>7</volume>(<issue>6</issue>):<fpage>533</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s42452-025-06991-6</pub-id>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Shi</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Weerawarna</surname> <given-names>P</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>K</given-names></string-name>, <string-name><surname>Richardson</surname> <given-names>T</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Building an explainable graph neural network by sparse learning for the drug-protein binding prediction</article-title>. <source>J Comput Biol</source>. <year>2025</year>;<volume>32</volume>(<issue>7</issue>):<fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi:<pub-id pub-id-type="doi">10.1101/2023.08.28.555203</pub-id>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Charoenkwan</surname> <given-names>P</given-names></string-name>, <string-name><surname>Schaduangrat</surname> <given-names>N</given-names></string-name>, <string-name><surname>Moni</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Shoombuatong</surname> <given-names>W</given-names></string-name>, <string-name><surname>Manavalan</surname> <given-names>B</given-names></string-name></person-group>. <article-title>Computational prediction and interpretation of druggable proteins using a stacked ensemble-learning framework</article-title>. <source>iScience</source>. <year>2022</year>;<volume>25</volume>(<issue>9</issue>):<fpage>104883</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.isci.2022.104883</pub-id>; <pub-id pub-id-type="pmid">36046193</pub-id></mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Altman</surname> <given-names>RB</given-names></string-name></person-group>. <article-title>Identifying druggable targets by protein microenvironments matching: application to transcription factors</article-title>. <source>CPT Pharmacomet Syst Pharmacol</source>. <year>2014</year>;<volume>3</volume>(<issue>1</issue>):<fpage>e93</fpage>. doi:<pub-id pub-id-type="doi">10.1038/psp.2013.66</pub-id>; <pub-id pub-id-type="pmid">24452614</pub-id></mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Owens</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Determining druggability</article-title>. <source>Nat Rev Drug Discov</source>. <year>2007</year>;<volume>6</volume>(<issue>3</issue>):<fpage>187</fpage>. doi:<pub-id pub-id-type="doi">10.1038/nrd2275</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Sakharkar</surname> <given-names>MK</given-names></string-name>, <string-name><surname>Sakharkar</surname> <given-names>KR</given-names></string-name>, <string-name><surname>Pervaiz</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Druggability of human disease genes</article-title>. <source>Int J Biochem Cell Biol</source>. <year>2007</year>;<volume>39</volume>(<issue>6</issue>):<fpage>1156</fpage>&#x2013;<lpage>64</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.biocel.2007.02.018</pub-id>; <pub-id pub-id-type="pmid">17446117</pub-id></mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Overington</surname> <given-names>JP</given-names></string-name>, <string-name><surname>Al-Lazikani</surname> <given-names>B</given-names></string-name>, <string-name><surname>Hopkins</surname> <given-names>AL</given-names></string-name></person-group>. <article-title>How many drug targets are there?</article-title> <source>Nat Rev Drug Discov</source>. <year>2006</year>;<volume>5</volume>(<issue>12</issue>):<fpage>993</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1038/nrd2199</pub-id>; <pub-id pub-id-type="pmid">17139284</pub-id></mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lindsay</surname> <given-names>MA</given-names></string-name></person-group>. <article-title>Finding new drug targets in the 21st century</article-title>. <source>Drug Discov Today</source>. <year>2005</year>;<volume>10</volume>(<issue>23&#x2013;24</issue>):<fpage>1683</fpage>&#x2013;<lpage>7</lpage>. doi:<pub-id pub-id-type="doi">10.1016/S1359-6446(05)03670-6</pub-id>; <pub-id pub-id-type="pmid">16376829</pub-id></mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cheng</surname> <given-names>AC</given-names></string-name>, <string-name><surname>Coleman</surname> <given-names>RG</given-names></string-name>, <string-name><surname>Smyth</surname> <given-names>KT</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Soulard</surname> <given-names>P</given-names></string-name>, <string-name><surname>Caffrey</surname> <given-names>DR</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Structure-based maximal affinity model predicts small-molecule druggability</article-title>. <source>Nat Biotechnol</source>. <year>2007</year>;<volume>25</volume>(<issue>1</issue>):<fpage>71</fpage>&#x2013;<lpage>5</lpage>. doi:<pub-id pub-id-type="doi">10.1038/nbt1273</pub-id>; <pub-id pub-id-type="pmid">17211405</pub-id></mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gashaw</surname> <given-names>I</given-names></string-name>, <string-name><surname>Ellinghaus</surname> <given-names>P</given-names></string-name>, <string-name><surname>Sommer</surname> <given-names>A</given-names></string-name>, <string-name><surname>Asadullah</surname> <given-names>K</given-names></string-name></person-group>. <article-title>What makes a good drug target?</article-title> <source>Drug Discov Today</source>. <year>2011</year>;<volume>16</volume>(<issue>23&#x2013;24</issue>):<fpage>1037</fpage>&#x2013;<lpage>43</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.drudis.2011.09.007</pub-id>; <pub-id pub-id-type="pmid">21945861</pub-id></mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Lai</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Prediction of potential drug targets based on simple sequence properties</article-title>. <source>BMC Bioinform</source>. <year>2007</year>;<volume>8</volume>:<fpage>353</fpage>. doi:<pub-id pub-id-type="doi">10.1186/1471-2105-8-353</pub-id>; <pub-id pub-id-type="pmid">17883836</pub-id></mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ding</surname> <given-names>H</given-names></string-name>, <string-name><surname>Takigawa</surname> <given-names>I</given-names></string-name>, <string-name><surname>Mamitsuka</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Similarity-based machine learning methods for predicting drug-target interactions: a brief review</article-title>. <source>Brief Bioinform</source>. <year>2014</year>;<volume>15</volume>(<issue>5</issue>):<fpage>734</fpage>&#x2013;<lpage>47</lpage>. doi:<pub-id pub-id-type="doi">10.1093/bib/bbt056</pub-id>; <pub-id pub-id-type="pmid">23933754</pub-id></mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Huang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Shang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>P</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Predict potential drug targets from the ion channel proteins based on SVM</article-title>. <source>J Theor Biol</source>. <year>2010</year>;<volume>262</volume>(<issue>4</issue>):<fpage>750</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.jtbi.2009.11.002</pub-id>; <pub-id pub-id-type="pmid">19903486</pub-id></mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Carter</surname> <given-names>AJ</given-names></string-name>, <string-name><surname>Kraemer</surname> <given-names>O</given-names></string-name>, <string-name><surname>Zwick</surname> <given-names>M</given-names></string-name>, <string-name><surname>Mueller-Fahrnow</surname> <given-names>A</given-names></string-name>, <string-name><surname>Arrowsmith</surname> <given-names>CH</given-names></string-name>, <string-name><surname>Edwards</surname> <given-names>AM</given-names></string-name></person-group>. <article-title>Target 2035: probing the human proteome</article-title>. <source>Drug Discov Today</source>. <year>2019</year>;<volume>24</volume>(<issue>11</issue>):<fpage>2111</fpage>&#x2013;<lpage>5</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.drudis.2019.06.020</pub-id>; <pub-id pub-id-type="pmid">31278990</pub-id></mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>M</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>F</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>T</given-names></string-name></person-group>. <article-title>DrugFinder: druggable protein identification model based on pre-trained models and evolutionary information</article-title>. <source>Algorithms</source>. <year>2023</year>;<volume>16</volume>(<issue>6</issue>):<fpage>263</fpage>. doi:<pub-id pub-id-type="doi">10.3390/a16060263</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ackloo</surname> <given-names>S</given-names></string-name>, <string-name><surname>Antolin</surname> <given-names>AA</given-names></string-name>, <string-name><surname>Bartolome</surname> <given-names>JM</given-names></string-name>, <string-name><surname>Beck</surname> <given-names>H</given-names></string-name>, <string-name><surname>Bullock</surname> <given-names>A</given-names></string-name>, <string-name><surname>Betz</surname> <given-names>UAK</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Target 2035&#x2014;an update on private sector contributions</article-title>. <source>RSC Med Chem</source>. <year>2023</year>;<volume>14</volume>(<issue>6</issue>):<fpage>1002</fpage>&#x2013;<lpage>11</lpage>. doi:<pub-id pub-id-type="doi">10.1039/D2MD00441K</pub-id>; <pub-id pub-id-type="pmid">37360399</pub-id></mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>McCloskey</surname> <given-names>K</given-names></string-name>, <string-name><surname>Sigel</surname> <given-names>EA</given-names></string-name>, <string-name><surname>Kearnes</surname> <given-names>S</given-names></string-name>, <string-name><surname>Xue</surname> <given-names>L</given-names></string-name>, <string-name><surname>Tian</surname> <given-names>X</given-names></string-name>, <string-name><surname>Moccia</surname> <given-names>D</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Machine learning on DNA-encoded libraries: a new paradigm for hit finding</article-title>. <source>J Med Chem</source>. <year>2020</year>;<volume>63</volume>(<issue>16</issue>):<fpage>8857</fpage>&#x2013;<lpage>66</lpage>. doi:<pub-id pub-id-type="doi">10.1021/acs.jmedchem.0c00452</pub-id>; <pub-id pub-id-type="pmid">32525674</pub-id></mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Torng</surname> <given-names>W</given-names></string-name>, <string-name><surname>Biancofiore</surname> <given-names>I</given-names></string-name>, <string-name><surname>Oehler</surname> <given-names>S</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Watson</surname> <given-names>I</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Deep learning approach for the discovery of tumor-targeting small organic ligands from DNA-encoded chemical libraries</article-title>. <source>ACS Omega</source>. <year>2023</year>;<volume>8</volume>(<issue>28</issue>):<fpage>25090</fpage>&#x2013;<lpage>100</lpage>. doi:<pub-id pub-id-type="doi">10.1021/acsomega.3c01775</pub-id>; <pub-id pub-id-type="pmid">37483198</pub-id></mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ahmad</surname> <given-names>S</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Feng</surname> <given-names>JA</given-names></string-name>, <string-name><surname>Hutchinson</surname> <given-names>A</given-names></string-name>, <string-name><surname>Zeng</surname> <given-names>H</given-names></string-name>, <string-name><surname>Ghiabi</surname> <given-names>P</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Discovery of a first-in-class small-molecule ligand for WDR91 using DNA-encoded chemical library selection followed by machine learning</article-title>. <source>J Med Chem</source>. <year>2023</year>;<volume>66</volume>(<issue>23</issue>):<fpage>16051</fpage>&#x2013;<lpage>61</lpage>. doi:<pub-id pub-id-type="doi">10.1021/acs.jmedchem.3c01471</pub-id>; <pub-id pub-id-type="pmid">37996079</pub-id></mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhou</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>SJ</given-names></string-name></person-group>. <article-title>Advances in machine-learning approaches to RNA-targeted drug design</article-title>. <source>Artif Intell Chem</source>. <year>2024</year>;<volume>2</volume>(<issue>1</issue>):<fpage>100053</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.aichem.2024.100053</pub-id>; <pub-id pub-id-type="pmid">38434217</pub-id></mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Volkamer</surname> <given-names>A</given-names></string-name>, <string-name><surname>Kuhn</surname> <given-names>D</given-names></string-name>, <string-name><surname>Grombacher</surname> <given-names>T</given-names></string-name>, <string-name><surname>Rippmann</surname> <given-names>F</given-names></string-name>, <string-name><surname>Rarey</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Combining global and local measures for structure-based druggability predictions</article-title>. <source>J Chem Inf Model</source>. <year>2012</year>;<volume>52</volume>(<issue>2</issue>):<fpage>360</fpage>&#x2013;<lpage>72</lpage>. doi:<pub-id pub-id-type="doi">10.1021/ci200454v</pub-id>; <pub-id pub-id-type="pmid">22148551</pub-id></mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dezs&#x0151;</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Ceccarelli</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Machine learning prediction of oncology drug targets based on protein and network properties</article-title>. <source>BMC Bioinform</source>. <year>2020</year>;<volume>21</volume>(<issue>1</issue>):<fpage>104</fpage>. doi:<pub-id pub-id-type="doi">10.1186/s12859-020-3442-9</pub-id>; <pub-id pub-id-type="pmid">32171238</pub-id></mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gong</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liao</surname> <given-names>B</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>P</given-names></string-name>, <string-name><surname>Zou</surname> <given-names>Q</given-names></string-name></person-group>. <article-title>DrugHybrid_BS: using hybrid feature combined with bagging-SVM to predict potentially druggable proteins</article-title>. <source>Front Pharmacol</source>. <year>2021</year>;<volume>12</volume>:<fpage>771808</fpage>. doi:<pub-id pub-id-type="doi">10.3389/fphar.2021.771808</pub-id>; <pub-id pub-id-type="pmid">34916947</pub-id></mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Sun</surname> <given-names>T</given-names></string-name>, <string-name><surname>Lai</surname> <given-names>L</given-names></string-name>, <string-name><surname>Pei</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Analysis of protein features and machine learning algorithms for prediction of druggable proteins</article-title>. <source>Quant Biol</source>. <year>2018</year>;<volume>6</volume>(<issue>4</issue>):<fpage>334</fpage>&#x2013;<lpage>43</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s40484-018-0157-2</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Xue</surname> <given-names>L</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>F</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Jing</surname> <given-names>R</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>J</given-names></string-name></person-group>. <article-title>The applications of deep learning algorithms on <italic>in silico</italic> druggable proteins identification</article-title>. <source>J Adv Res</source>. <year>2022</year>;<volume>41</volume>:<fpage>219</fpage>&#x2013;<lpage>31</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.jare.2022.01.009</pub-id>; <pub-id pub-id-type="pmid">36328750</pub-id></mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Butcher</surname> <given-names>SP</given-names></string-name></person-group>. <article-title>Target discovery and validation in the post-genomic era</article-title>. <source>Neurochem Res</source>. <year>2003</year>;<volume>28</volume>(<issue>2</issue>):<fpage>367</fpage>&#x2013;<lpage>71</lpage>. doi:<pub-id pub-id-type="doi">10.1023/a:1022349805831</pub-id>; <pub-id pub-id-type="pmid">12608710</pub-id></mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>M&#x00FC;ller</surname> <given-names>S</given-names></string-name>, <string-name><surname>Ackloo</surname> <given-names>S</given-names></string-name>, <string-name><surname>Al Chawaf</surname> <given-names>A</given-names></string-name>, <string-name><surname>Al-Lazikani</surname> <given-names>B</given-names></string-name>, <string-name><surname>Antolin</surname> <given-names>A</given-names></string-name>, <string-name><surname>Baell</surname> <given-names>JB</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Target 2035&#x2014;update on the quest for a probe for every protein</article-title>. <source>RSC Med Chem</source>. <year>2022</year>;<volume>13</volume>(<issue>1</issue>):<fpage>13</fpage>&#x2013;<lpage>21</lpage>. doi:<pub-id pub-id-type="doi">10.1039/d1md00228g</pub-id>; <pub-id pub-id-type="pmid">35211674</pub-id></mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Jamali</surname> <given-names>AA</given-names></string-name>, <string-name><surname>Ferdousi</surname> <given-names>R</given-names></string-name>, <string-name><surname>Razzaghi</surname> <given-names>S</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Safdari</surname> <given-names>R</given-names></string-name>, <string-name><surname>Ebrahimie</surname> <given-names>E</given-names></string-name></person-group>. <article-title>DrugMiner: comparative analysis of machine learning algorithms for prediction of potential druggable proteins</article-title>. <source>Drug Discov Today</source>. <year>2016</year>;<volume>21</volume>(<issue>5</issue>):<fpage>718</fpage>&#x2013;<lpage>24</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.drudis.2016.01.007</pub-id>; <pub-id pub-id-type="pmid">26821132</pub-id></mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bakheet</surname> <given-names>TM</given-names></string-name>, <string-name><surname>Doig</surname> <given-names>AJ</given-names></string-name></person-group>. <article-title>Properties and identification of human protein drug targets</article-title>. <source>Bioinformatics</source>. <year>2009</year>;<volume>25</volume>(<issue>4</issue>):<fpage>451</fpage>&#x2013;<lpage>7</lpage>. doi:<pub-id pub-id-type="doi">10.1093/bioinformatics/btp002</pub-id>; <pub-id pub-id-type="pmid">19164304</pub-id></mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bhasin</surname> <given-names>M</given-names></string-name>, <string-name><surname>Raghava</surname> <given-names>GPS</given-names></string-name></person-group>. <article-title>ESLpred: SVM-based method for subcellular localization of eukaryotic proteins using dipeptide composition and PSI-BLAST</article-title>. <source>Nucleic Acids Res</source>. <year>2004</year>;<volume>32</volume>:<fpage>W414</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1093/nar/gkh350</pub-id>; <pub-id pub-id-type="pmid">15215421</pub-id></mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dhanda</surname> <given-names>SK</given-names></string-name>, <string-name><surname>Vir</surname> <given-names>P</given-names></string-name>, <string-name><surname>Raghava</surname> <given-names>GPS</given-names></string-name></person-group>. <article-title>Designing of interferon-gamma inducing MHC class-II binders</article-title>. <source>Biol Direct</source>. <year>2013</year>;<volume>8</volume>:<fpage>30</fpage>. doi:<pub-id pub-id-type="doi">10.1186/1745-6150-8-30</pub-id>; <pub-id pub-id-type="pmid">24304645</pub-id></mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Saravanan</surname> <given-names>V</given-names></string-name>, <string-name><surname>Gautham</surname> <given-names>N</given-names></string-name></person-group>. <article-title>Harnessing computational biology for exact linear B-cell epitope prediction: a novel amino acid composition-based feature descriptor</article-title>. <source>OMICS</source>. <year>2015</year>;<volume>19</volume>(<issue>10</issue>):<fpage>648</fpage>&#x2013;<lpage>58</lpage>. doi:<pub-id pub-id-type="doi">10.1089/omi.2015.0095</pub-id>; <pub-id pub-id-type="pmid">26406767</pub-id></mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>P</given-names></string-name>, <string-name><surname>Li</surname> <given-names>F</given-names></string-name>, <string-name><surname>Leier</surname> <given-names>A</given-names></string-name>, <string-name><surname>Marquez-Lago</surname> <given-names>TT</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>iFeature: a Python package and web server for features extraction and selection from protein and peptide sequences</article-title>. <source>Bioinformatics</source>. <year>2018</year>;<volume>34</volume>(<issue>14</issue>):<fpage>2499</fpage>&#x2013;<lpage>502</lpage>. doi:<pub-id pub-id-type="doi">10.1093/bioinformatics/bty140</pub-id>; <pub-id pub-id-type="pmid">29528364</pub-id></mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lee</surname> <given-names>TY</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>ZQ</given-names></string-name>, <string-name><surname>Hsieh</surname> <given-names>SJ</given-names></string-name>, <string-name><surname>Breta&#x00F1;a</surname> <given-names>NA</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>CT</given-names></string-name></person-group>. <article-title>Exploiting maximal dependence decomposition to identify conserved motifs from a group of aligned signal sequences</article-title>. <source>Bioinformatics</source>. <year>2011</year>;<volume>27</volume>(<issue>13</issue>):<fpage>1780</fpage>&#x2013;<lpage>7</lpage>. doi:<pub-id pub-id-type="doi">10.1093/bioinformatics/btr291</pub-id>; <pub-id pub-id-type="pmid">21551145</pub-id></mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yao</surname> <given-names>L</given-names></string-name>, <string-name><surname>Rzhetsky</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Quantitative systems-level determinants of human genes targeted by successful drugs</article-title>. <source>Genome Res</source>. <year>2008</year>;<volume>18</volume>(<issue>2</issue>):<fpage>206</fpage>&#x2013;<lpage>13</lpage>. doi:<pub-id pub-id-type="doi">10.1101/gr.6888208</pub-id>; <pub-id pub-id-type="pmid">18083776</pub-id></mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Y&#x0131;ld&#x0131;r&#x0131;m</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Goh</surname> <given-names>KI</given-names></string-name>, <string-name><surname>Cusick</surname> <given-names>ME</given-names></string-name>, <string-name><surname>Barab&#x00E1;si</surname> <given-names>AL</given-names></string-name>, <string-name><surname>Vidal</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Drug-target network</article-title>. <source>Nat Biotechnol</source>. <year>2007</year>;<volume>25</volume>(<issue>10</issue>):<fpage>1119</fpage>&#x2013;<lpage>26</lpage>. doi:<pub-id pub-id-type="doi">10.1038/nbt1338</pub-id>; <pub-id pub-id-type="pmid">17921997</pub-id></mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cao</surname> <given-names>DS</given-names></string-name>, <string-name><surname>Xiao</surname> <given-names>N</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>QS</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>AF</given-names></string-name></person-group>. <article-title>Rcpi: R/Bioconductor package to generate various descriptors of proteins, compounds and their interactions</article-title>. <source>Bioinformatics</source>. <year>2015</year>;<volume>31</volume>(<issue>2</issue>):<fpage>279</fpage>&#x2013;<lpage>81</lpage>. doi:<pub-id pub-id-type="doi">10.1093/bioinformatics/btu624</pub-id>; <pub-id pub-id-type="pmid">25246429</pub-id></mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hao</surname> <given-names>J</given-names></string-name>, <string-name><surname>Ho</surname> <given-names>TK</given-names></string-name></person-group>. <article-title>Machine learning made easy: a review of <italic>Scikit-learn</italic> package in Python programming language</article-title>. <source>J Educ Behav Stat</source>. <year>2019</year>;<volume>44</volume>(<issue>3</issue>):<fpage>348</fpage>&#x2013;<lpage>61</lpage>. doi:<pub-id pub-id-type="doi">10.3102/1076998619832248</pub-id>.</mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Breiman</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Bagging predictors</article-title>. <source>Mach Learn</source>. <year>1996</year>;<volume>24</volume>(<issue>2</issue>):<fpage>123</fpage>&#x2013;<lpage>40</lpage>. doi:<pub-id pub-id-type="doi">10.1007/BF00058655</pub-id>.</mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Breiman</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Random forests</article-title>. <source>Mach Learn</source>. <year>2001</year>;<volume>45</volume>(<issue>1</issue>):<fpage>5</fpage>&#x2013;<lpage>32</lpage>. doi:<pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id>.</mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hughes</surname> <given-names>G</given-names></string-name></person-group>. <article-title>On the mean accuracy of statistical pattern recognizers</article-title>. <source>IEEE Trans Inf Theory</source>. <year>1968</year>;<volume>14</volume>(<issue>1</issue>):<fpage>55</fpage>&#x2013;<lpage>63</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TIT.1968.1054102</pub-id>.</mixed-citation></ref>
<ref id="ref-46"><label>[46]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Friedman</surname> <given-names>JH</given-names></string-name></person-group>. <article-title>Stochastic gradient boosting</article-title>. <source>Comput Stat Data Anal</source>. <year>2002</year>;<volume>38</volume>(<issue>4</issue>):<fpage>367</fpage>&#x2013;<lpage>78</lpage>. doi:<pub-id pub-id-type="doi">10.1016/S0167-9473(01)00065-2</pub-id>.</mixed-citation></ref>
<ref id="ref-47"><label>[47]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>T</given-names></string-name>, <string-name><surname>Guestrin</surname> <given-names>C</given-names></string-name></person-group>. <article-title>XGBoost: a scalable tree boosting system</article-title>. In: <conf-name>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name>; <year>2016 Aug 13&#x2013;17</year>; <publisher-loc>San Francisco, CA, USA</publisher-loc>. p. <fpage>785</fpage>&#x2013;<lpage>94</lpage>. doi:<pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id>.</mixed-citation></ref>
<ref id="ref-48"><label>[48]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Brewka</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Artificial intelligence&#x2014;a modern approach by Stuart Russell and Peter Norvig, Prentice Hall. Series in artificial intelligence, Englewood Cliffs, NJ</article-title>. <source>Knowl Eng Rev</source>. <year>1996</year>;<volume>11</volume>(<issue>1</issue>):<fpage>78</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1017/s0269888900007724</pub-id>.</mixed-citation></ref>
<ref id="ref-49"><label>[49]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cover</surname> <given-names>T</given-names></string-name>, <string-name><surname>Hart</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Nearest neighbor pattern classification</article-title>. <source>IEEE Trans Inf Theory</source>. <year>1967</year>;<volume>13</volume>(<issue>1</issue>):<fpage>21</fpage>&#x2013;<lpage>7</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TIT.1967.1053964</pub-id>.</mixed-citation></ref>
<ref id="ref-50"><label>[50]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Swain</surname> <given-names>PH</given-names></string-name>, <string-name><surname>Hauska</surname> <given-names>H</given-names></string-name></person-group>. <article-title>The decision tree classifier: design and potential</article-title>. <source>IEEE Trans Geosci Electron</source>. <year>1977</year>;<volume>15</volume>(<issue>3</issue>):<fpage>142</fpage>&#x2013;<lpage>7</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TGE.1977.6498972</pub-id>.</mixed-citation></ref>
<ref id="ref-51"><label>[51]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>L&#x00F3;pez-Cort&#x00E9;s</surname> <given-names>A</given-names></string-name>, <string-name><surname>Paz-y-Mi&#x00F1;o</surname> <given-names>C</given-names></string-name>, <string-name><surname>Guerrero</surname> <given-names>S</given-names></string-name>, <string-name><surname>Cabrera-Andrade</surname> <given-names>A</given-names></string-name>, <string-name><surname>Barigye</surname> <given-names>SJ</given-names></string-name>, <string-name><surname>Munteanu</surname> <given-names>CR</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>OncoOmics approaches to reveal essential genes in breast cancer: a panoramic view from pathogenesis to precision medicine</article-title>. <source>Sci Rep</source>. <year>2020</year>;<volume>10</volume>:<fpage>5285</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-020-62279-2</pub-id>; <pub-id pub-id-type="pmid">32210335</pub-id></mixed-citation></ref>
<ref id="ref-52"><label>[52]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Repana</surname> <given-names>D</given-names></string-name>, <string-name><surname>Nulsen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Dressler</surname> <given-names>L</given-names></string-name>, <string-name><surname>Bortolomeazzi</surname> <given-names>M</given-names></string-name>, <string-name><surname>Venkata</surname> <given-names>SK</given-names></string-name>, <string-name><surname>Tourna</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>The network of cancer genes (NCG): a comprehensive catalogue of known and candidate cancer genes from cancer sequencing screens</article-title>. <source>Genome Biol</source>. <year>2019</year>;<volume>20</volume>(<issue>1</issue>):<fpage>1</fpage>. doi:<pub-id pub-id-type="doi">10.1186/s13059-018-1612-0</pub-id>; <pub-id pub-id-type="pmid">30606230</pub-id></mixed-citation></ref>
<ref id="ref-53"><label>[53]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Daanouni</surname> <given-names>O</given-names></string-name>, <string-name><surname>Cherradi</surname> <given-names>B</given-names></string-name>, <string-name><surname>Tmiri</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Predicting diabetes diseases using mixed data and supervised machine learning algorithms</article-title>. In: <conf-name>Proceedings of the 4th International Conference on Smart City Applications; 2019 Oct 2&#x2013;4</conf-name>; <publisher-loc>Casablanca, Morocco</publisher-loc>. p. <fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3368756.3369072</pub-id>.</mixed-citation></ref>
<ref id="ref-54"><label>[54]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Nai-Arun</surname> <given-names>N</given-names></string-name>, <string-name><surname>Sittidech</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Ensemble learning model for diabetes classification</article-title>. <source>Adv Mater Res</source>. <year>2014</year>;<volume>931-932</volume>:<fpage>1427</fpage>&#x2013;<lpage>31</lpage>. doi:<pub-id pub-id-type="doi">10.4028/www.scientific.net/amr.931-932.1427</pub-id>.</mixed-citation></ref>
<ref id="ref-55"><label>[55]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Mung</surname> <given-names>PS</given-names></string-name>, <string-name><surname>Phyu</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Ensemble learning method for enhancing healthcare classification</article-title>. In: <conf-name>Proceedings of 2020 the 10th International Workshop on Computer Science and Engineering WCSE 2020; 2020 Jun 19&#x2013;21</conf-name>; <publisher-loc>Shanghai, China</publisher-loc>. doi:<pub-id pub-id-type="doi">10.18178/wcse.2020.02.024</pub-id>.</mixed-citation></ref>
<ref id="ref-56"><label>[56]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Guerrero</surname> <given-names>S</given-names></string-name>, <string-name><surname>L&#x00F3;pez-Cort&#x00E9;s</surname> <given-names>A</given-names></string-name>, <string-name><surname>Indacochea</surname> <given-names>A</given-names></string-name>, <string-name><surname>Garc&#x00ED;a-C&#x00E1;rdenas</surname> <given-names>JM</given-names></string-name>, <string-name><surname>Zambrano</surname> <given-names>AK</given-names></string-name>, <string-name><surname>Cabrera-Andrade</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Analysis of racial/ethnic representation in select basic and applied cancer research studies</article-title>. <source>Sci Rep</source>. <year>2018</year>;<volume>8</volume>:<fpage>13978</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-018-32264-x</pub-id>; <pub-id pub-id-type="pmid">30228363</pub-id></mixed-citation></ref>
<ref id="ref-57"><label>[57]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>L&#x00F3;pez-Cort&#x00E9;s</surname> <given-names>A</given-names></string-name>, <string-name><surname>Paz-Y-Mi&#x00F1;o</surname> <given-names>C</given-names></string-name>, <string-name><surname>Guerrero</surname> <given-names>S</given-names></string-name>, <string-name><surname>Jaramillo-Koupermann</surname> <given-names>G</given-names></string-name>, <string-name><surname>Le&#x00F3;n C&#x00E1;ceres</surname> <given-names>&#x00C1;</given-names></string-name>, <string-name><surname>Intriago-Balde&#x00F3;n</surname> <given-names>DP</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Pharmacogenomics, biomarker network, and allele frequencies in colorectal cancer</article-title>. <source>Pharmacogenom J</source>. <year>2020</year>;<volume>20</volume>(<issue>1</issue>):<fpage>136</fpage>&#x2013;<lpage>58</lpage>. doi:<pub-id pub-id-type="doi">10.1038/s41397-019-0102-4</pub-id>; <pub-id pub-id-type="pmid">31616044</pub-id></mixed-citation></ref>
<ref id="ref-58"><label>[58]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Rodr&#x00ED;guez-P&#x00E9;rez</surname> <given-names>R</given-names></string-name>, <string-name><surname>Bajorath</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Interpretation of machine learning models using shapley values: application to compound potency and multi-target activity predictions</article-title>. <source>J Comput Aided Mol Des</source>. <year>2020</year>;<volume>34</volume>(<issue>10</issue>):<fpage>1013</fpage>&#x2013;<lpage>26</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10822-020-00314-0</pub-id>; <pub-id pub-id-type="pmid">32361862</pub-id></mixed-citation></ref>
<ref id="ref-59"><label>[59]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Xu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Li</surname> <given-names>D</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>X</given-names></string-name></person-group>. <article-title>A t-SNE based classification approach to compositional microbiome data</article-title>. <source>Front Genet</source>. <year>2020</year>;<volume>11</volume>:<fpage>620143</fpage>. doi:<pub-id pub-id-type="doi">10.3389/fgene.2020.620143</pub-id>; <pub-id pub-id-type="pmid">33381156</pub-id></mixed-citation></ref>
<ref id="ref-60"><label>[60]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mayer</surname> <given-names>J</given-names></string-name>, <string-name><surname>Jin</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wurster</surname> <given-names>TH</given-names></string-name>, <string-name><surname>Makowski</surname> <given-names>MR</given-names></string-name>, <string-name><surname>Kolbitsch</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Evaluation of synergistic image registration for motion-corrected coronary NaF-PET-MR</article-title>. <source>Philos Trans A Math Phys Eng Sci</source>. <year>2021</year>;<volume>379</volume>(<issue>2200</issue>):<fpage>20200202</fpage>. doi:<pub-id pub-id-type="doi">10.1098/rsta.2020.0202</pub-id>; <pub-id pub-id-type="pmid">33966463</pub-id></mixed-citation></ref>
</ref-list>
</back></article>













