<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CSSE</journal-id>
<journal-id journal-id-type="nlm-ta">CSSE</journal-id>
<journal-id journal-id-type="publisher-id">CSSE</journal-id>
<journal-title-group>
<journal-title>Computer Systems Science &#x0026; Engineering</journal-title>
</journal-title-group>
<issn pub-type="ppub">0267-6192</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">34373</article-id>
<article-id pub-id-type="doi">10.32604/csse.2023.034373</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>A Novel Framework for Learning and Classifying the Imbalanced Multi-Label Data</article-title>
<alt-title alt-title-type="left-running-head">A Novel Framework for Learning and Classifying the Imbalanced Multi-Label Data</alt-title>
<alt-title alt-title-type="right-running-head">A Novel Framework for Learning and Classifying the Imbalanced Multi-Label Data</alt-title>
</title-group>	
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Chitra</surname><given-names>P. K. A.</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Balamurugan</surname><given-names>S. Appavu alias</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Geetha</surname><given-names>S.</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Kadry</surname><given-names>Seifedine</given-names></name><xref ref-type="aff" rid="aff-4">4</xref><xref ref-type="aff" rid="aff-5">5</xref><xref ref-type="aff" rid="aff-6">6</xref></contrib>
<contrib id="author-5" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Kim</surname><given-names>Jungeun</given-names></name><xref ref-type="aff" rid="aff-7">7</xref><email>jekim@kongju.ac.kr</email></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Han</surname><given-names>Keejun</given-names></name><xref ref-type="aff" rid="aff-8">8</xref></contrib>
<aff id="aff-1"><label>1</label><institution>Department of Computer Science and Engineering, SRM Institute of Science and Technology</institution>, <addr-line>Tiruchirappalli, Tamil Nadu, 603203</addr-line>, <country>India</country></aff>
<aff id="aff-2"><label>2</label><institution>Department of Computer Science and Engineering, Periyar Maniammai Institute of Science &#x0026; Technology (Deemed to be University)</institution>, <addr-line>Thanjavur, Tamil Nadu, 613403</addr-line>, <country>India</country></aff>
<aff id="aff-3"><label>3</label><institution>School of Computer Science and Engineering</institution>, <addr-line>Chennai, Tamil Nadu, 600048</addr-line>, <country>India</country></aff>
<aff id="aff-4"><label>4</label><institution>Department of Applied Data Science, Noroff University College</institution>, <addr-line>Kristiansand, 4612</addr-line>, <country>Norway</country></aff>
<aff id="aff-5"><label>5</label><institution>Artificial Intelligence Research Center (AIRC), College of Engineering and Information Technology, Ajman University</institution>, <addr-line>P.O. Box 346, Ajman</addr-line>, <country>United Arab Emirates</country></aff>
<aff id="aff-6"><label>6</label><institution>Department of Electrical and Computer Engineering, Lebanese American University</institution>, <addr-line>Byblos, 10150</addr-line>, <country>Lebanon</country></aff>
<aff id="aff-7"><label>7</label><institution>Department of Software, Kongju National University</institution>, <addr-line>Cheonan, 31080</addr-line>, <country>Republic of Korea</country></aff>
<aff id="aff-8"><label>8</label><institution>Division of Computer Engineering, Hansung University</institution>, <addr-line>Seoul, 02876</addr-line>, <country>Republic of Korea</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Jungeun Kim. Email: <email>jekim@kongju.ac.kr</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2024</year></pub-date>
<pub-date date-type="pub" publication-format="electronic"><day>13</day><month>09</month><year>2024</year></pub-date>
<volume>48</volume>
<issue>5</issue>
<fpage>1367</fpage>
<lpage>1385</lpage>
<history>
<date date-type="received">
<day>15</day>
<month>7</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>14</day>
<month>12</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2024 The Authors.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CSSE_34373.pdf"></self-uri>
<abstract>
<p>A generalization of supervised single-label learning based on the assumption that each sample in a dataset may belong to more than one class simultaneously is called multi-label learning. The main objective of this work is to create a novel framework for learning and classifying imbalanced multi-label data. This work proposes a framework of two phases. The imbalanced distribution of the multi-label dataset is addressed through the proposed Borderline MLSMOTE resampling method in phase 1. Later, an adaptive weighted <italic>l</italic><sub>21</sub> norm regularized (Elastic-net) multi-label logistic regression is used to predict unseen samples in phase 2. The proposed Borderline MLSMOTE resampling method focuses on samples with concurrent high labels in contrast to conventional MLSMOTE. The minority labels in these samples are called difficult minority labels and are more prone to penalize classification performance. The concurrent measure is considered borderline, and labels associated with samples are regarded as borderline labels in the decision boundary. In phase II, a novel adaptive <italic>l</italic><sub>21</sub> norm regularized weighted multi-label logistic regression is used to handle balanced data with different weighted synthetic samples. Experimentation on various benchmark datasets shows the outperformance of the proposed method and its powerful predictive performances over existing conventional state-of-the-art multi-label methods.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Multi-label imbalanced data</kwd>
<kwd>multi-label learning</kwd>
<kwd>Borderline MLSMOTE</kwd>
<kwd>concurrent multi-label</kwd>
<kwd>adaptive weighted multi-label elastic net</kwd>
<kwd>difficult minority label</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>An essential variation of typical supervised learning is known as multi-label learning. Unlike traditional supervised learning, the labels in multi-label learning are not mutually exclusive. However, it might be associated. In the case of multi-label learning, every example relates to multiple class labels concurrently [<xref ref-type="bibr" rid="ref-1">1</xref>]. Each data sample is represented predictor vector associated with multiple labels simultaneously. Let <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2264;</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x2264;</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2264;</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x2264;</mml:mo><mml:mi>k</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula> be the training data of&#x2014;labelmulti-label, which consists of n features and k labels in the label space. Single instance (<italic>x</italic><sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub>) in a multi-label data represents an <italic>n</italic>-dimensional predictive vector <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> of real values and <italic>y</italic><sub>i</sub> is the associated <italic>k</italic>-dimensional label space <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> of binary values, <italic>y</italic><sub>ij</sub> &#x003D; 1 indicates label <italic>j</italic> is present in the label set of <italic>x</italic><sub><italic>i </italic>,</sub> else <italic>y</italic><sub>ij</sub> &#x003D; 0. Given training data set <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2264;</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x2264;</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2264;</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x2264;</mml:mo><mml:mi>k</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula>, the multi-label learning role is to figure out a function <italic>f(.): X <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mo stretchy="false">&#x2192;</mml:mo></mml:math></inline-formula> 2</italic><sup><italic>y</italic></sup> which predicts a correct label set <italic>y <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mo>&#x2286;</mml:mo></mml:math></inline-formula>Y</italic> for a provided undetected multi-label example instance. In many real-world applications, multi-label learning is commonly used, which includes automatic image annotation [<xref ref-type="bibr" rid="ref-2">2</xref>], text categorization [<xref ref-type="bibr" rid="ref-3">3</xref>], bioinformatics [<xref ref-type="bibr" rid="ref-4">4</xref>], music categorization [<xref ref-type="bibr" rid="ref-5">5</xref>], and drug side effect prediction [<xref ref-type="bibr" rid="ref-6">6</xref>]. For the past decade, the research in multi-label learning developed various multi-label learning approaches. Still, many challenges exist in multi-learning. The number of labels for prediction, which is exponential in size, is considered a fundamental challenge to multi-label data. This involves exploiting label correlation among labels, over-fitting due to high dimensional predictive space, and highly imbalanced training sets.</p>
<p>Most modern existing multi-label methods try to address the first problem generally, and a couple of efforts are made to resolve the second and third issues. The performance of many standard learning algorithms is degraded due to the class imbalance problem [<xref ref-type="bibr" rid="ref-7">7</xref>]. This is the case where the classes are not present equally. The number of positive examples for each category is less than its negative counterparts. This may lead to performance degradation of the learning method. As the learning algorithms are biased to prefer the majority class, the imbalanced data can negatively affect the learning algorithms [<xref ref-type="bibr" rid="ref-8">8</xref>]. With the neglect of imbalanced class distribution, conventional state-of-the-art learning algorithms perform poorly and produce unsatisfactory suboptimal results [<xref ref-type="bibr" rid="ref-9">9</xref>,<xref ref-type="bibr" rid="ref-10">10</xref>]. The irrelevant classes can be well identified with the poor-performing learning algorithms, while the minority is reversed. The class imbalance ratio of the existing data sets used for this research has been mentioned in [<xref ref-type="bibr" rid="ref-11">11</xref>]. The use of ensemble techniques to improve accuracy in single-label classifiers has been discussed in [<xref ref-type="bibr" rid="ref-12">12</xref>]. The easiest way to handle imbalanced data is through sampling. The sampling methods are of two types: (1) under-sampling; (2) over-sampling. Random sampling leads to overfitting if the sampling ratios are not appropriately set [<xref ref-type="bibr" rid="ref-13">13</xref>]. The synthetic minority over-sampling procedure (<italic>SMOTE</italic>) makes synthetic minority cases through interpolation amongst proper training samples and k-nearest neighborhoods. Many enhancements were done over <italic>SMOTE</italic> [<xref ref-type="bibr" rid="ref-14">14</xref>&#x2013;<xref ref-type="bibr" rid="ref-16">16</xref>] to improve the performance of the learning process for imbalanced binary and multi-class data.</p>
<p><italic>SMOTE</italic> was adopted by [<xref ref-type="bibr" rid="ref-17">17</xref>] to handle imbalance conditions in multi-label data. First, the instances with minority labels are selected as the seed, and their nearest neighbors are identified. Next, the features of the artificial instances are created based on randomly chosen neighbors. Then the synthetic instances are created with the feature values and the label information obtained from samples with minority labels and neighbors of minority labels. This process works with the k-Nearest Neighbor (<italic>kNN</italic>) method to find the nearest neighbors for a minority label. Finally, distances are computed based on Euclidean measure, and as a result, static <italic>k</italic> numbers of neighbors are returned for a minority label. Regression is a simple but powerful statistical method to discover the linear and non-linear relationships between predictors and the response variable. Logistic Regression (<italic>LR</italic>) is a robust and computationally fast discriminative method designed to model and capture each class&#x2019;s posterior probabilities. <italic>LR</italic> is a conventional statistical technique addressing binary problems. However, <italic>LR</italic> over-fits the training samples when the number of predictors vastly outstrips the number of samples. The elastic net can carry out automatic variable selection and continual contraction all at once and chooses the group of associated variables. This elastic net serves especially when the number of predictors is much larger than the variety of observations in the sample data.</p>
<p>Machine learning and AI methods have started to show their dominance in many fields, reference [<xref ref-type="bibr" rid="ref-18">18</xref>] showed the usage of the above two methods in the field of handwritten alphabet recognition, which plays a crucial role in pattern recognition, computer vision, and image processing. Deep Learning has been a boon in automated effective image processing. Deep learning-based automated weed in crops was presented by [<xref ref-type="bibr" rid="ref-19">19</xref>]. They used ten various rabi crops for their experimentation and proved the use of deep learning in the field. AI and deep learning-based methods could be used to handle the sheer amount of data that is being generated nowadays, and it was addressed by [<xref ref-type="bibr" rid="ref-20">20</xref>]. They managed high-dimensional data and explored their research in various application areas to justify the dominance of machine learning and deep learning. Deep learning and machine learning could also be used to address real-time social needs and have been presented by [<xref ref-type="bibr" rid="ref-21">21</xref>] to automatically detect garbage areas in remote locations.</p>
<p>This work addresses the imbalanced characteristic of logistic regression and the extension of logistic regression to penalized multi-label logistic regression in this paper. The adaptive weighted elastic net is used in the second phase to handle the synthetic samples produced in the first phase. This work offers an approach to make the logistic regression model work on imbalanced data. This framework introduces a new pre-processing method called <italic>Borderline MLSMOTE</italic> in phase 1. The newly created balanced multi-label dataset will combine the training dataset and generated synthetic samples. One noteworthy feature of the offered pre-processing approach is that it assists in expanding the minority labels in areas where the concurrent appearance of the minority and majority labels is too high. In the second phase, the <italic>l</italic><sub>21</sub>-norm regularized weighted logistic regression (adaptive elastic-net) is used to handle the over-fitting and variable selection simultaneously to make the learning and prediction over the balanced data. The objectives of this research are:
<list list-type="bullet">
<list-item>
<p>A framework of two phases to handle and predict imbalanced multi-label data has been presented.</p></list-item>
<list-item>
<p><italic>Borderline MLSMOTE</italic> has been introduced to handle difficult concurrent minority labels.</p></list-item>
<list-item>
<p>Adaptive weighted <italic>l</italic><sub>21</sub> norm regularization is presented and introduced to handle the problem of overfitting and variable selection in high-dimensional multi-label data.</p></list-item>
<list-item>
<p>To conduct in-depth experiments on eighteen benchmark multi-label datasets to demonstrate that the proposed framework will handle imbalanced data more effectively than existing multi-label learning techniques.</p></list-item>
</list></p>
<sec id="s1_1">
<label>1.1</label>
<title>Organization of the Paper</title>
<p><xref ref-type="sec" rid="s1">Section 1</xref> describes the introduction of multi-label learning, imbalanced data, logistic regression, and the paper&#x2019;s contributions. <xref ref-type="sec" rid="s2">Section 2</xref> presents related works in imbalanced learning, multi-label learning, and logistic regression. <xref ref-type="sec" rid="s3">Section 3</xref> discusses the background of multi-label learning, measures for imbalance in multi-label data, a review of logistic regression, and the proposed system. <xref ref-type="sec" rid="s4">Section 4</xref> provides the speculative setup needed for the construction of the paper. <xref ref-type="sec" rid="s5">Section 5</xref> defines outcomes as well as discussion. Finally, Section 6 concludes with the end and future improvement of the work addressed in this paper.</p>
</sec>
</sec>
<sec id="s2">
<label>2</label>
<title>Methods</title>
<p>Before the suggested frame was introduced, a few simple understandings regarding the methods were outlined briefly. Inside the frame, the <italic>Border MLSMOTE</italic> sampling technique is used at the pre-processing data level to resolve the imbalanced dataset by creating a synthetic dataset. Then adaptive <italic>l</italic><sub>21</sub>-norm regularised multi-label logistic regression is used to predict the balanced multi-label data. Finally, the results show the significance of the proposed framework.</p>
<sec id="s2_1">
<label>2.1</label>
<title>Multi-Label Learning</title>
<p>Let <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2264;</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x2264;</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2264;</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x2264;</mml:mo><mml:mi>k</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula> be the multi-label training data with <italic>p</italic> observations. The <italic>i</italic><sup>th</sup> multi-label sample instance (<italic>x</italic><sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub>), x<sub>i</sub> is an <italic>n</italic>-dimensional predictive vector <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> of real values, and <italic>y</italic><sub>i</sub> is the associated <italic>k</italic>-dimensional label space <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> of binary values. For an unseen sample <italic>x</italic>, the classifier <italic>f</italic>(.) predicts <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> as a label vector for the unseen sample <italic>x</italic>. Classification whereby each example could be connected with an assortment of class labels. Each label comprises just a binary value. The multi-label classification maps an example instance to multiple labels. It is a generalization of supervised single label learning which map an example instance q<sub>i</sub> <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:mo>&#x2208;</mml:mo></mml:math></inline-formula> {Q} to a label set <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12">
<mml:mmultiscripts>
<mml:mo>&#x20AC;</mml:mo>
<mml:none/> 
<mml:none/> 
<mml:mprescripts/>
<mml:none/>
<mml:mi>&#x2113;</mml:mi>
</mml:mmultiscripts>
<mml:mspace width="thinmathspace" />
<mml:mo fence="false" stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">&#x00A3;</mml:mi>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:math>
</inline-formula> and is given in <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref>.</p>
<p><disp-formula id="eqn-1">
<label>(1)</label>
<mml:math id="mml-eqn-1" display="block"><mml:mi>&#x210F;</mml:mi><mml:mo>:</mml:mo><mml:mi>&#x03C7;</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo><mml:msup><mml:mn>2</mml:mn><mml:mrow><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:mrow><mml:mtext>i.e.,</mml:mtext></mml:mrow><mml:mspace width="2em" /><mml:mi>M</mml:mi><mml:mi>L</mml:mi><mml:mi>C</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>&#x210F;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>:</mml:mo><mml:mrow><mml:mtext>&#xA0;qi&#xA0;</mml:mtext></mml:mrow><mml:mo stretchy="false">&#x21A6;</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mi>&#x2113;</mml:mi><mml:mspace width="1em" /><mml:mrow><mml:mtext>&#xA0;where</mml:mtext></mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mo>|</mml:mo><mml:mi>&#x2113;</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mo>&#x2265;</mml:mo><mml:mn>2.</mml:mn></mml:math></disp-formula></p>
<p>Problem transformation alters multi-labeled data into single labels; conventional single-label classification approaches are used on transformed data. The transformed single-labeled data are binary classifiers; conventional single-label classifiers are enough to produce the model. Adaptation in the algorithm in multi-label learning frees single-labeled classifiers to alter them to choose multi-labeled data. Thus, algorithm variation strategies are effective and free from information loss.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Measures of Imbalance in Multi-Label Data</title>
<p>Multi-label classification is more complex than single-label classification when addressing class imbalance. IRperLabel and MeanIR were presented by [<xref ref-type="bibr" rid="ref-22">22</xref>] to measure imbalance in multi-label data. The IRperLabel (Imbalance Rate per Label) measures each label in the dataset. It provides individual imbalance levels in the dataset. <xref ref-type="disp-formula" rid="eqn-2">Eq. (2)</xref> shows the definition of IRperLabel:</p>
<p><disp-formula id="eqn-2">
<label>(2)</label>
<mml:math id="mml-eqn-2" display="block"><mml:mrow><mml:mtext>IRperLabel&#xA0;</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mrow><mml:mtext mathvariant="italic">argmax</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x2113;</mml:mi><mml:mi>&#x03B5;</mml:mi><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:munderover><mml:mi>h</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x2113;</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:munderover><mml:mi>h</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x2113;</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mrow><mml:mtext>&#xA0;where&#xA0;</mml:mtext></mml:mrow><mml:mi>h</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x2113;</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mtext>&#x00A0;</mml:mtext><mml:mi>&#x2113;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>&#x03B5;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn><mml:mtext>&#x00A0;</mml:mtext><mml:mi>&#x2113;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>&#x03B5;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable><mml:mo>}</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>MeanIR represents the average imbalance in multi-label data and is shown in <xref ref-type="disp-formula" rid="eqn-3">Eq. (3)</xref>. It shows the average of <italic>IRperLabel</italic> for all labels.</p>
<p><disp-formula id="eqn-3">
<label>(3)</label>
<mml:math id="mml-eqn-3" display="block"><mml:mrow><mml:mtext>MeanIR&#xA0;</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mtext mathvariant="italic">IRperLabel</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>&#x2113;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<p>Apart from imbalance, concurrence among the imbalanced labels also needs to be considered before balancing. Concurrence among labels represents joint appearances of relevant and irrelevant labels of the same instance. This is measured using SCUMBLE (Score of ConcUrrence among iMBalanced LabEls). It concerns the amount of imbalance variance among relevant and irrelevant labels of each instance. The concurrence measure of each instance (SCUMBLE) in the dataset is calculated, and the average of all the instances <italic>SCUMBLE<sub>i</sub></italic> measure is given in <xref ref-type="disp-formula" rid="eqn-4">Eqs. (4)</xref> and <xref ref-type="disp-formula" rid="eqn-5">(5)</xref>.</p>
<p><disp-formula id="eqn-4">
<label>(4)</label>
<mml:math id="mml-eqn-4" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="italic">SCUMBLE</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mo>|</mml:mo><mml:mi>D</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>D</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:munderover><mml:msub><mml:mrow><mml:mtext>IRperLabel</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mo movablelimits="false">&#x220F;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>D</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:msub><mml:mrow><mml:mtext mathvariant="italic">IRperLabel</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p><disp-formula id="eqn-5">
<label>(5)</label>
<mml:math id="mml-eqn-5" display="block"><mml:mrow><mml:mtext>SCUMBLE&#xA0;</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>D</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mo>|</mml:mo><mml:mi>D</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:msub><mml:mrow><mml:mtext mathvariant="italic">SCUMBLE</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></disp-formula></p>
<p>In multi-label data, a group of positive or negative labels might exist, which necessitates the algorithms to be designed in such a way as to handle a group of labels instead of a single one. Label concurrence of yeast data is depicted in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>. Arc represents labels in the dataset. The length segment of the arc describes the number of instances associated with each label. Minority labels in yeast data are class 14, class 9, class 10, and class 11; these labels appear together with one or more irrelevant labels, and these four relevant labels are difficult. <xref ref-type="fig" rid="fig-2">Fig. 2</xref> shows labels concurrence of emotional data. This dataset contains six labels, and the picture indicates the number of samples related to each label and interactions among labels. For example, angry-aggressive is a minority label in the emotions dataset. It appears together with other majority labels like relaxing-calm and sad-lonely.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Label concurrence view of yeast data</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34373-fig-1.tif"/>
</fig><fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Label concurrence view of emotional data</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34373-fig-2.tif"/>
</fig>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Logistic Regression and Elastic Net</title>
<p>The logistic regression (LR) learns the relationship of predictive variables to binary 1 (&#x201C;success&#x201D;/&#x201C;Presence&#x201D;) or 0 (&#x201C;failure&#x201D;/&#x201C;Absence&#x201D;) valued response variables. When the response data is binary, logistic regression is used to find the relationship between the predictor and responses. The LR extracts some weighted predictors from predictor space and then combines them linearly. The conditional probability of LR is the joint prediction of labels in the label space given as in <xref ref-type="disp-formula" rid="eqn-6">Eqs. (6)</xref> and <xref ref-type="disp-formula" rid="eqn-7">(7)</xref>.<disp-formula id="eqn-6">
<label>(6)</label>
<mml:math id="mml-eqn-6" display="block"><mml:mrow><mml:mi mathvariant="bold-italic">p</mml:mi><mml:mi mathvariant="bold-italic">P</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="bold-italic">L</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mi mathvariant="bold-italic">X</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi mathvariant="bold-italic">L</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="normal">&#x0398;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mo movablelimits="false">&#x220F;</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mo>=</mml:mo><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">k</mml:mi></mml:mrow></mml:msubsup><mml:mi mathvariant="bold-italic">P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>&#x2113;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi mathvariant="bold-italic">X</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<p><disp-formula id="eqn-7">
<label>(7)</label>
<mml:math id="mml-eqn-7" display="block"><mml:mo>=</mml:mo><mml:msubsup><mml:mo movablelimits="false">&#x220F;</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">k</mml:mi></mml:mrow></mml:msubsup><mml:mfrac><mml:msup><mml:mi mathvariant="bold-italic">e</mml:mi><mml:mrow><mml:mrow><mml:msub><mml:mi>&#x2113;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:msubsup><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">T</mml:mi></mml:mrow></mml:msubsup><mml:mi mathvariant="bold-italic">X</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:msup><mml:mi mathvariant="bold-italic">e</mml:mi><mml:mrow><mml:msubsup><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">T</mml:mi></mml:mrow></mml:msubsup><mml:mi mathvariant="bold-italic">X</mml:mi></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi mathvariant="bold-italic">e</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msubsup><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">T</mml:mi></mml:mrow></mml:msubsup><mml:mi mathvariant="bold-italic">X</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:math></disp-formula>where <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> are the coefficient for the <italic>i</italic><sup>th</sup> logistic regression to be determined and <italic>&#x03B2;</italic><sub>0</sub> is the intercept; <italic>X</italic> &#x003D; (<italic>x</italic><sub>i1</sub>, <italic>x</italic><sub>i2</sub>, ..., <italic>x</italic><sub>ik</sub>), <italic>i</italic> &#x003D; 1, 2, 3, ..., k is the input predictors of the training instances; <italic>Y</italic> &#x003D; (<italic>y</italic><sub>i1</sub>, <italic>y</italic><sub>i2</sub>, ...., <italic>y</italic><sub>im</sub>) is the regression object vectors, i.e., labels/targets. The predictor variables are continuous, while the response variable is binary (yes/no or present /absent). The regression coefficients are estimated using maximum likelihood estimation. The negative log-likelihood of <italic>k</italic> observations is as in <xref ref-type="disp-formula" rid="eqn-8">Eq. (8)</xref> under the independence assumption.</p>
<p><disp-formula id="eqn-8">
<label>(8)</label>
<mml:math id="mml-eqn-8" display="block"><mml:mi mathvariant="bold-italic">P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="bold-italic">L</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mi mathvariant="bold-italic">X</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi mathvariant="bold-italic">L</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="normal">&#x0398;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mfrac><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow><mml:mi mathvariant="bold-italic">k</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">k</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>&#x2113;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mi mathvariant="bold-italic">f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msup><mml:mi mathvariant="bold-italic">X</mml:mi><mml:mrow><mml:mi mathvariant="bold">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>&#x2113;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">l</mml:mi><mml:mi mathvariant="bold-italic">o</mml:mi><mml:mi mathvariant="bold-italic">g</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi mathvariant="bold-italic">f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msup><mml:mi mathvariant="bold-italic">X</mml:mi><mml:mrow><mml:mi mathvariant="bold">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>To handle the ill-condition and the over-fitting problems, <xref ref-type="disp-formula" rid="eqn-8">Eq. (8)</xref> is added with a penalty term. The penalized function of the logistic regression is given in <xref ref-type="disp-formula" rid="eqn-9">Eq. (9)</xref>. The penalty term is defined in <xref ref-type="disp-formula" rid="eqn-10">Eq. (10)</xref>.</p>
<p><disp-formula id="eqn-9">
<label>(9)</label>
<mml:math id="mml-eqn-9" display="block"><mml:mi mathvariant="bold-italic">P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="bold-italic">L</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mi mathvariant="bold-italic">X</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold-italic">a</mml:mi><mml:mi mathvariant="bold-italic">r</mml:mi><mml:mi mathvariant="bold-italic">g</mml:mi><mml:mi mathvariant="bold-italic">m</mml:mi><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mi mathvariant="bold-italic">n</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x0398;</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mi mathvariant="bold-italic">L</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="normal">&#x0398;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi mathvariant="bold-italic">R</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="normal">&#x0398;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<p><disp-formula id="eqn-10">
<label>(10)</label>
<mml:math id="mml-eqn-10" display="block"><mml:mi mathvariant="bold-italic">L</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="normal">&#x0398;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mfrac><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow><mml:mi mathvariant="bold-italic">k</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">k</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">l</mml:mi><mml:mi mathvariant="bold-italic">o</mml:mi><mml:mi mathvariant="bold-italic">g</mml:mi></mml:mrow><mml:mspace width="thinmathspace" /><mml:mi mathvariant="bold-italic">p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>&#x2113;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">l</mml:mi><mml:mi mathvariant="bold-italic">o</mml:mi><mml:mi mathvariant="bold-italic">g</mml:mi></mml:mrow><mml:mspace width="thinmathspace" /><mml:mi mathvariant="bold-italic">p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">&#x2113;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo></mml:math></disp-formula></p>
<p><disp-formula id="eqn-11">
<label>(11)</label>
<mml:math id="mml-eqn-11" display="block"><mml:mi mathvariant="bold-italic">R</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="normal">&#x0398;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow><mml:mo>+</mml:mo><mml:mfrac><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mn mathvariant="bold">2</mml:mn></mml:mrow></mml:mrow></mml:msub><mml:mi mathvariant="bold-italic">m</mml:mi></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="bold-italic">a</mml:mi><mml:mi mathvariant="bold-italic">r</mml:mi><mml:mi mathvariant="bold-italic">g</mml:mi><mml:mi mathvariant="bold-italic">m</mml:mi><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mi mathvariant="bold-italic">n</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03B2;</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">m</mml:mi></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mo symmetric="true">&#x2016;</mml:mo><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">y</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mo symmetric="true">&#x2016;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mn mathvariant="bold">2</mml:mn></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mn mathvariant="bold">2</mml:mn></mml:mrow></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mn mathvariant="bold">2</mml:mn></mml:mrow></mml:mrow></mml:msub><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mo>&#x003C;</mml:mo><mml:mi mathvariant="bold-italic">m</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mo symmetric="true">&#x2016;</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mo symmetric="true">&#x2016;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mn mathvariant="bold">2</mml:mn></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mn mathvariant="bold">2</mml:mn></mml:mrow></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow></mml:mrow></mml:msub><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mo>&#x003C;</mml:mo><mml:mi mathvariant="bold-italic">m</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mo symmetric="true">&#x2016;</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:mo symmetric="true">&#x2016;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow></mml:mrow></mml:msup></mml:math></disp-formula></p>
<p><inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mi>R</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="normal">&#x0398;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> in <xref ref-type="disp-formula" rid="eqn-11">Eq. (11)</xref> is the elastic net regularization that learns a linear model to predict the response vector from the predictor by minimizing the squared loss with &#x2113;<sub>2</sub> regularization and &#x2113;<sub>1</sub> norm constraint. <italic>&#x03BB;</italic><sub><italic>1</italic></sub>, <italic>&#x03BB;</italic><sub><italic>2</italic></sub> are the hyperparameters that decide the trade-off between the term regularization and error. These hyperparameters are the tuning parameters that provide the goodness of fit and the complexity of the model. An Elastic net is a mixture of <italic>l</italic><sub>1</sub> (lasso) and <italic>l</italic><sub>2</sub> (ridge) penalties. The lasso part of the elastic net performs variable selection, while the ridge part in the penalization stabilizes the solution paths. The amount of predictor variables exceeds the range of all samples/observations in the case of high-dimensional data analysis. When <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mrow><mml:mn mathvariant="bold">2</mml:mn></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula> &#x003D; 0, elastic net reduces to lasso. Elastic net is effective when dimensional space <italic>n</italic> is much greater than the observations <italic>p</italic> (<italic>p</italic> &#x003E; <italic>n</italic>). In our research, the predictors greatly exceed the samples as they increase with the label space. Thus we choose to develop a statistical model based on the elastic net for multi-label data processing. The elastic net gives computational flexibility and performance stability.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Proposed Algorithm</title>
<p>CAL500, Enron, Corel5k, Rcv1 (subset1), Rcv1 (subset2), Mediamill, tmc2007, Corel6k, and eurlex-sm have SCUMBLE values more than 0.1, and they are extremely difficult MLDs. They take a high degree of concurrence amongst labels with distinct imbalance levels. Existing sampling methods &#x2018;won&#x2019;t be suitable for handling such labels with high concurrence values. On the other hand, emotions, Medical, Scene, Slashdot, Bibtex, and Genbase, have low SCUMBLE values, and the imbalanced processing with the conventional methods over these datasets will benefit. The proposed framework uses a two-stage approach to balance and predicts the imbalanced multi-label data. The first stage uses the pre-processing data to balance the imbalanced multi-label dataset. The multi-label SMOTE (<italic>MLSMOTE</italic>) implemented is modified to treat labels in the training set as a one-<italic>vs.</italic>-rest way to create new synthetic samples. Then a new balanced multi-label dataset of original and synthetic samples is given as input for adaptive weighted <italic>l</italic><sub>21</sub><italic>-</italic>norm regularized logistic regression to make the learning and prediction over the balanced dataset. <xref ref-type="fig" rid="fig-3">Fig. 3</xref> describes the flow of the data balancing approach.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Phase 1-balancing the data set through <italic>Borderline-MLSMOTE</italic></title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34373-fig-3.tif"/>
</fig>
<sec id="s2_4_1">
<label>2.4.1</label>
<title>Borderline MLSMOTE</title>
<p>SMOTE system uses heuristics to pick samples with minority labels; ergo, their operation was better than other pre-processing approaches. However, to accomplish a better forecast, the sample minority labels that are high interactions with all majority labels must be obtained instead of sample majority labels. These minority labels using high concurrent values are considered borderline, and the neighboring ones tend to be more inclined to become misclassified compared to the main ones, much against the uncontrollable ones. Therefore, those concurrent minority labels tend to be more crucial for classification.</p>
<p>The instances with those concurrent labels tend to be somewhat more prone to become misclassified. Focusing resampling on those samples with labels that are concurrent makes it more beneficial compared to doing whole minority labels. However, the samples from the samples that are concurrent can contribute little to this classification. Our approaches are all predicated on the synthetic minority over-sampling Technique. The synthetic sample creation pre-processing creates synthetic minority samples to oversample the concurrent minority class. For each single concurrent minority label, its own k closest nearest neighbor is calculated, and then some samples are randomly chosen in line with this sampling speed. Now, the brand-new synthetic samples have been generated concerned with concurrent minority labels along with their own chosen nearest neighbor. Unlike the multi-label SMOTE system, our suggested pre-processing only reinforces the borderline (concurrent) minority samples. The generated synthetic samples are subsequently added to the unique training set. The most widely selected neighborhood size k &#x003D; 5 [<xref ref-type="bibr" rid="ref-23">23</xref>] is used in this work. The flow of the proposed Borderline MLSMOTE pre-processing is shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>.</p>
</sec>
<sec id="s2_4_2">
<label>2.4.2</label>
<title>Adaptive Weighted Logistic Regression for Multi-Label Data</title>
<p>The adaptive weighted elastic net is used to learn and predict the balanced data set created in the first phase. The weighted regularisation ensures the handling of synthetic samples created in phase 1. The adaptive elastic-net guarantees variable selection by adding <italic>l</italic><sub>2</sub> regularization with an adaptive lasso to address multi-collinearity problems. Reference [<xref ref-type="bibr" rid="ref-24">24</xref>] outperformed well than an adaptive lasso and elastic&#x2013;net in terms of accuracy while maintaining a higher value of true positive rate and a lower value of false negative rate with the selected predictors. The elastic net proposed in [<xref ref-type="bibr" rid="ref-25">25</xref>] is modified to an adaptive weighted elastic net via logistic regression to address three issues: over-fitting, biased estimation, multi-collinearity, and low false-positive rate. The regularization part of the elastic-net logistic regression is modified by adding weight terms to the lasso and ridge parts. The <xref ref-type="disp-formula" rid="eqn-12">Eq. (12)</xref> is a modified adaptive elastic net.</p>
<p><disp-formula id="eqn-12">
<label>(12)</label>
<mml:math id="mml-eqn-12" display="block"><mml:mi>A</mml:mi><mml:mi>d</mml:mi><mml:mi>a</mml:mi><mml:mi>R</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="normal">&#x0398;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mfrac><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mi>m</mml:mi></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mtext mathvariant="italic">argmin</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x03B2;</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mo symmetric="true">&#x2016;</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo symmetric="true">&#x2016;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:mfrac><mml:mi>m</mml:mi><mml:mn>2</mml:mn></mml:mfrac><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x003C;</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mo symmetric="true">&#x2016;</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo symmetric="true">&#x2016;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:mi>m</mml:mi><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x003C;</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:msup><mml:mrow><mml:mo symmetric="true">&#x2016;</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo symmetric="true">&#x2016;</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></disp-formula>where <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> &#x003E; 0, <italic>i</italic> &#x003D; 1, 2, ..., <italic>m</italic> are the weighted penalty coefficients. The key idea in adaptive elastic net is the weight parameter only. This weight makes the adaptive elastic net perform different amounts of shrinkage to different predictors and penalizes the smaller coefficient predictors more severely. The coefficient estimations for the elastic net are done first to construct the adaptive weights of the lasso and ridge part of the adaptive elastic net. This is presented in <xref ref-type="disp-formula" rid="eqn-13">Eq. (13)</xref>.</p>
<p><disp-formula id="eqn-13">
<label>(13)</label>
<mml:math id="mml-eqn-13" display="block"><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold-italic">R</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="bold">&#x0398;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B3;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x2026;</mml:mo><mml:mi>K</mml:mi></mml:math></disp-formula></p>
<p>&#x03B3; is a positive constant. In this paper, we use &#x03B3; &#x003D; 1. The <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mi mathvariant="bold-italic">R</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="bold">&#x0398;</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> in <xref ref-type="disp-formula" rid="eqn-11">Eq. (11)</xref> gives the initial estimator for weight in <xref ref-type="disp-formula" rid="eqn-13">Eq. (13)</xref>. The learning through the adaptive weighted elastic net is presented in <xref ref-type="fig" rid="fig-4">Fig. 4</xref>. The proposed framework is illustrated in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Phase 2-learning of balanced multi-label data through the adaptive weighted elastic net</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34373-fig-4.tif"/>
</fig><fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Proposed framework</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34373-fig-5.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Experimental Setup</title>
<p>This section describes the list of multi-label datasets employed with this experimentation and the evaluation metrics used to evaluate the learning algorithms.</p>
<sec id="s3_1">
<label>3.1</label>
<title>Data Sets Used</title>
<p>The datasets used for the experimentation are taken from various domains like music, text, image, video, and biology. All the benchmark data used here are available in the MULAN data repository. The data sets are shown in <xref ref-type="table" rid="table-1">Table 1</xref>.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Description of the data sets used</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>S. No.</th>
<th>Data Set</th>
<th>Domain</th>
<th>#n</th>
<th>#d</th>
<th>#l</th>
<th>L<sub>C</sub></th>
<th>L<sub>D</sub></th>
<th>IR<sub>min</sub></th>
<th>IR<sub>avg</sub></th>
<th>IR<sub>max</sub></th>
<th>SCUMBLE</th>
</tr>
</thead>
<tbody>
<tr>
<td>1.</td>
<td>Corel5k</td>
<td>Image</td>
<td>5000</td>
<td>499</td>
<td>44</td>
<td>2.214</td>
<td>0.050</td>
<td>3.460</td>
<td>17.857</td>
<td>50.000</td>
<td>0.393</td>
</tr>
<tr>
<td>2.</td>
<td>Mediamill</td>
<td>Video</td>
<td>43907</td>
<td>120</td>
<td>29</td>
<td>4.010</td>
<td>0.138</td>
<td>1.748</td>
<td>7.092</td>
<td>45.455</td>
<td>0.355</td>
</tr>
<tr>
<td>3.</td>
<td>CAL500</td>
<td>Music</td>
<td>502</td>
<td>68</td>
<td>174</td>
<td>25.058</td>
<td>0.202</td>
<td>1.040</td>
<td>3.846</td>
<td>24.390</td>
<td>0.336</td>
</tr>
<tr>
<td>4.</td>
<td>Enron</td>
<td>Text</td>
<td>1702</td>
<td>1001</td>
<td>53</td>
<td>3.378</td>
<td>0.130</td>
<td>1.00</td>
<td>5.348</td>
<td>43.478</td>
<td>0.302</td>
</tr>
<tr>
<td>5.</td>
<td>Corel16k</td>
<td>Image</td>
<td>13766</td>
<td>500</td>
<td>161</td>
<td>2.867</td>
<td>0.018</td>
<td>1.808</td>
<td>34.1552</td>
<td>126.80</td>
<td>0.279</td>
</tr>
<tr>
<td>6.</td>
<td>Rcv1(subset 1)</td>
<td>Image</td>
<td>6000</td>
<td>472</td>
<td>42</td>
<td>2.458</td>
<td>0.059</td>
<td>3.344</td>
<td>15.152</td>
<td>50.000</td>
<td>0.223</td>
</tr>
<tr>
<td>7.</td>
<td>Rcv1(subset 2)</td>
<td>Image</td>
<td>6000</td>
<td>472</td>
<td>39</td>
<td>2.170</td>
<td>0.056</td>
<td>3.215</td>
<td>15.873</td>
<td>47.619</td>
<td>0.209</td>
</tr>
<tr>
<td>8.</td>
<td>Eurlex-sm</td>
<td>Text</td>
<td>19348</td>
<td>250</td>
<td>27</td>
<td>1.492</td>
<td>0.055</td>
<td>1.447</td>
<td>5.848</td>
<td>34.483</td>
<td>0.182</td>
</tr>
<tr>
<td>9.</td>
<td>tmc2007</td>
<td>Text</td>
<td>28596</td>
<td>500</td>
<td>15</td>
<td>2.100</td>
<td>0.140</td>
<td>1.447</td>
<td>5.848</td>
<td>34.483</td>
<td>0.175</td>
</tr>
<tr>
<td>10.</td>
<td>Yeast</td>
<td>Biology</td>
<td>2417</td>
<td>103</td>
<td>13</td>
<td>4.233</td>
<td>0.325</td>
<td>1.328</td>
<td>2.778</td>
<td>12.500</td>
<td>0.104</td>
</tr>
<tr>
<td>11.</td>
<td>Bibtex</td>
<td>Text</td>
<td>7395</td>
<td>1836</td>
<td>159</td>
<td>2.402</td>
<td>0.015</td>
<td>0.450</td>
<td>12.4983</td>
<td>20.4314</td>
<td>0.094</td>
</tr>
<tr>
<td>12.</td>
<td>Medical</td>
<td>Text</td>
<td>978</td>
<td>1449</td>
<td>45</td>
<td>1.275</td>
<td>0.077</td>
<td>2.674</td>
<td>11.236</td>
<td>43.478</td>
<td>0.046</td>
</tr>
<tr>
<td>13.</td>
<td>Genbase</td>
<td>biology</td>
<td>662</td>
<td>1186</td>
<td>27</td>
<td>1.252</td>
<td>0.046</td>
<td>1.4494</td>
<td>37.3146</td>
<td>171.00</td>
<td>0.028</td>
</tr>
<tr>
<td>14.</td>
<td>Slashdot</td>
<td>Text</td>
<td>3782</td>
<td>53</td>
<td>14</td>
<td>1.134</td>
<td>0.081</td>
<td>5.464</td>
<td>10.989</td>
<td>35.714</td>
<td>0.013</td>
</tr>
<tr>
<td>15.</td>
<td>Emotions</td>
<td>Music</td>
<td>593</td>
<td>72</td>
<td>6</td>
<td>1.869</td>
<td>0.311</td>
<td>1.247</td>
<td>2.146</td>
<td>3.003</td>
<td>0.010</td>
</tr>
<tr>
<td>16.</td>
<td>Scene</td>
<td>image</td>
<td>2407</td>
<td>294</td>
<td>6</td>
<td>1.074</td>
<td>0.179</td>
<td>3.521</td>
<td>4.566</td>
<td>5.618</td>
<td>0.0003</td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn><p>Note: #n&#x2013;Number of samples, #d&#x2013;Number of features, #l&#x2013;Number of labels, L<sub>C</sub>&#x2013;Label Cardinality, L<sub>D</sub>&#x2013;Label Density, IR<sub>min</sub>&#x2013;Imbalance Ratio minimum, IR<sub>max</sub>&#x2013;Imbalance Ratio maximum, IR<sub>avg</sub>&#x2013;Imbalance Ratio average.</p></fn></table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Evaluation Metrics</title>
<p>The efficiency assessment of multi-label techniques will be far more ambitious than single-label classification since it involves several labels.</p>
<p>1) <italic>Hamming Loss &#x2193;</italic> requires the error of prediction, overlooking missing errors into consideration, and testimonials the typical example-label set misclassification. Therefore, a lower <italic>h_loss</italic> value shows higher classifier performance. The hamming_Loss is specified as in <xref ref-type="disp-formula" rid="eqn-14">Eq. (14)</xref>.</p>
<p><disp-formula id="eqn-14">
<label>(14)</label>
<mml:math id="mml-eqn-14" display="block"><mml:mi>h</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>q</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>q</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mo>&#x0142;</mml:mo><mml:mrow></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:mi>h</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>|</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>Or</p>
<p>Let <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mi>t</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:mi>t</mml:mi><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:mi>f</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:mi>f</mml:mi><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are the count of true positive, true negative, false positive, and false negative of ith sample, respectively, and the <italic>h_loss</italic> can be defined as in <xref ref-type="disp-formula" rid="eqn-15">Eq. (15)</xref>.</p>
<p><disp-formula id="eqn-15">
<label>(15)</label>
<mml:math id="mml-eqn-15" display="block"><mml:mi>h</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>q</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>q</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mfrac><mml:mrow><mml:mi>f</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>f</mml:mi><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>f</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>t</mml:mi><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>f</mml:mi><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
<p>2) <italic>One_error &#x2193;</italic> outlines the lack of high-positioned labels <italic>vs.</italic> the proper label in the instant collection. This step chooses the good value between 1 and 0. The smaller the value of one_error, the classifier does effectively, and it is defined as in <xref ref-type="disp-formula" rid="eqn-16">Eq. (16)</xref>.</p>
<p><disp-formula id="eqn-16">
<label>(16)</label>
<mml:math id="mml-eqn-16" display="block"><mml:mi>O</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mrow><mml:mtext mathvariant="italic">error</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x210F;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo symmetric="true">&#x2016;</mml:mo><mml:munder><mml:mrow><mml:mtext mathvariant="italic">argmax</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x03BB;</mml:mi><mml:mi>&#x03B5;</mml:mi><mml:mi>&#x2113;</mml:mi></mml:mrow></mml:munder><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo symmetric="true">&#x2016;</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>This step is comparable to the classification error just in a single-label classification problem.</p>
<p>3) <italic>Ranking Loss &#x2193;</italic> defines the quality of reversely ordered label sets for the specified example, plus it is also as in <xref ref-type="disp-formula" rid="eqn-17">Eq. (17)</xref>.</p>
<p><disp-formula id="eqn-17">
<label>(17)</label>
<mml:math id="mml-eqn-17" display="block"><mml:mrow><mml:mi>R</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>k</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>g</mml:mi></mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mover><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup><mml:mo accent="false">&#x00AF;</mml:mo></mml:mover><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula></p><p>When the ranking loss is smaller, the learning algorithm&#x2019;s performance is better.</p>
<p>4) <italic>Average Precision &#x2191;</italic> computes the normal percentage of proper labels in every label set. Fundamentally, the quality has performed all applicable labels. This is provided from <xref ref-type="disp-formula" rid="eqn-18">Eq. (18)</xref>.</p>
<p><disp-formula id="eqn-18">
<label>(18)</label>
<mml:math id="mml-eqn-18" display="block"><mml:mrow><mml:mi>A</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mtext mathvariant="italic">Precision</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mo>|</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mo>&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x03BB;</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:msup><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x2208;</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mrow><mml:mtext>&#xA0;where&#xA0;</mml:mtext></mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<p>Better the significance of moderate precision improved the learning algorithm&#x2019;s performance, and when standard precision &#x003D; 1, the learning algorithm shows optimum performance.</p>
<p>5) <italic>Subset_Accuracy &#x2191;</italic> is characterized by the Jaccard similarity coefficient amongst label sets <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:mi>&#x210F;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> and <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. This step will be a type across all instances and is represented as in <xref ref-type="disp-formula" rid="eqn-19">Eq. (19)</xref>.</p>
<p><disp-formula id="eqn-19">
<label>(19)</label>
<mml:math id="mml-eqn-19" display="block"><mml:mrow><mml:mtext>Subset</mml:mtext></mml:mrow><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x210F;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:mi>&#x210F;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2229;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>&#x210F;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x222A;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
<p>6) <italic>Coverage &#x2193;</italic> is the portion of covered labels in the instance collection. Tiny the present, the higher the performance, i.e., far better label coverage. We must use the example set to cover the remaining uncovered labels if this measure is high. This is given in <xref ref-type="disp-formula" rid="eqn-20">Eq. (20)</xref>.</p>
<p><disp-formula id="eqn-20">
<label>(20)</label>
<mml:math id="mml-eqn-20" display="block"><mml:mrow><mml:mtext mathvariant="italic">Cover</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x210F;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:munder><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mrow><mml:mtext>rank</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x03BB;</mml:mi><mml:mi>&#x03B5;</mml:mi><mml:mi>&#x2113;</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:math></disp-formula></p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Results and Discussion</title>
<sec id="s4_1">
<label>4.1</label>
<title>Performance Comparison of Proposed Method against Competing Methods</title>
<p>The performance of a multi-label classifier is assessed in the shape of several test metrics. These classification results are evaluated with five multi-label measures: Hamming Loss, Subset Accuracy, Average Precision, One Error, Ranking Loss, and Coverage. The Hamming Loss is a sample-based step that assesses the gaps between the predicted and the provided label set. Lower the Hamming, the greater the predictions. Average Precision is an example-based step and a usual performance metric. Finally, One_Error and Ranking Loss are ranking-based metrics. The different comparison methods are chosen because they have relatively high performance and efficiency.</p>
<p><xref ref-type="table" rid="table-2">Table 2</xref> presents the overall performance of the proposed system. <xref ref-type="table" rid="table-2">Table 2</xref> shows that the suggested framework works quite competing methods utilized for the experimentation. The proposed method performs well on most data sets and is like other rival techniques with hamming loss and accuracy. The CAL500 exhibits 92% subset accuracy and 99% precision. The hamming loss for this dataset is 0.1. The Emotions and Medical datasets&#x2019; accuracy ranges are 87% and 93 %, respectively. The precision value on &#x2018;Emotion&#x2019;s data reaches a good 99%. In Medical data, the misclassification rate was reduced to 0.03 only. The Corel5k is a dataset with a high concurrence problem as its SCUMBLE is 0.39. Therefore, the accuracy is 83%. But for the same dataset, the MLSR produces 89% accuracy. However, the Precision is 93% with the proposed framework, and the hamming loss is 0.01. For Rcv1 (subset1) and Rcv1 (subset 2) data, the subset accuracy is 97% and 80%, respectively. The Hamming Loss measures 0.09 for Rcv1 (subset1) and 0.01 for Rcv1 (subset2). The hamming loss for Rcv1 (subset2) is much less than Rcv1 (subset1) as the SCUMBLE value of Rcv1 (subset2) is less than Rcv1 (subse1). This shows that the dataset with low concurrent value makes the learner perform well. The hamming loss for the Scene dataset is 0.02 with 99% accuracy and 92% precision value. The Scene data has a low SCUMBLE value, i.e., 0.003, among other datasets taken for our experimentation. Even though many datasets got benefited from the proposed framework, the Core5k gives the second-best Hamming Loss of 0.1007. The MLSR performs best with the Corel5k dataset, resulting in 0.1005 Hamming Loss and 89% accuracy. But still, the proposed system with Corek5k gives the second-best 83% accuracy. In general, the dataset with high concurrence labels is much benefited from the proposed system. Finally, experimentation on 17 datasets shows that the proposed framework performs better than the other nine popular methods.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Performance comparison of the proposed method against competing methods</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Data set</th>
<th>Performance measures</th>
<th>IBLR</th>
<th>ECC</th>
<th>BR</th>
<th>RAkEL</th>
<th>ML&#x2013;kNN</th>
<th>CLR</th>
<th>MLSR</th>
<th>MDDM</th>
<th>MORP</th>
<th>Proposed</th>
</tr>
</thead>
<tbody>
<tr>
<td>CAL500</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.1420</td>
<td>0.2015</td>
<td>0.2518</td>
<td>0.2566</td>
<td>0.2198</td>
<td>0.1987</td>
<td>0.1928</td>
<td>0.1638</td>
<td>0.1637</td>
<td><bold>0.1036</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.5601</td>
<td>0.5564</td>
<td>0.5089</td>
<td>0.4743</td>
<td>0.5331</td>
<td>0.5598</td>
<td>0.5038</td>
<td>0.5172</td>
<td>0.5728</td>
<td><bold>0.9283</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.7538</td>
<td>0.7843</td>
<td>0.7580</td>
<td>0.6992</td>
<td>0.7698</td>
<td>0.7999</td>
<td>0.6982</td>
<td>0.6538</td>
<td>0.6628</td>
<td><bold>0.9937</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.1677</td>
<td>0.2037</td>
<td>0.2208</td>
<td>0.3006</td>
<td>0.2096</td>
<td>0.1776</td>
<td>0.2293</td>
<td>0.2182</td>
<td>0.2273</td>
<td><bold>0.1272</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.1423</td>
<td>0.2749</td>
<td>0.2306</td>
<td>0.2980</td>
<td>0.2053</td>
<td>0.2699</td>
<td>0.2335</td>
<td>0.2284</td>
<td>0.2102</td>
<td><bold>0.1087</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>2.0169</td>
<td>2.0607</td>
<td>2.1063</td>
<td>2.5465</td>
<td>2.0623</td>
<td>1.8922</td>
<td>1.8920</td>
<td>1.9872</td>
<td>2.0371</td>
<td><bold>1.5269</bold></td>
</tr>
<tr>
<td>Emotions</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.2010</td>
<td>0.1993</td>
<td>0.2555</td>
<td>0.2566</td>
<td>0.2246</td>
<td>0.2125</td>
<td>0.2802</td>
<td>0.2173</td>
<td>0.2302</td>
<td><bold>0.1566</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.5390</td>
<td>0.5175</td>
<td>0.5207</td>
<td>0.4743</td>
<td>0.5073</td>
<td>0.5196</td>
<td>0.4338</td>
<td>0.5544</td>
<td>0.5114</td>
<td><bold>0.8743</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.6947</td>
<td>0.7338</td>
<td>0.7624</td>
<td>0.6992</td>
<td>0.7885</td>
<td>0.7819</td>
<td>0.6517</td>
<td>0.3630</td>
<td>0.7169</td>
<td><bold>0.9992</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.3015</td>
<td>0.2706</td>
<td>0.1967</td>
<td>0.3006</td>
<td><bold>0.1713</bold></td>
<td>0.1819</td>
<td>0.3545</td>
<td>0.7706</td>
<td>0.2709</td>
<td>0.2006</td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.3930</td>
<td>0.3070</td>
<td>0.3576</td>
<td>0.3980</td>
<td>0.2969</td>
<td>0.2985</td>
<td>0.4772</td>
<td>0.8819</td>
<td>0.4014</td>
<td><bold>0.2980</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.5869</td>
<td>2.4199</td>
<td>1.9360</td>
<td>2.5465</td>
<td>1.8449</td>
<td>1.9022</td>
<td>2.7284</td>
<td>4.3086</td>
<td>2.3542</td>
<td><bold>1.0465</bold></td>
</tr>
<tr>
<td>Medical</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.0774</td>
<td>0.0658</td>
<td>0.0729</td>
<td>0.0629</td>
<td>0.0573</td>
<td>0.0728</td>
<td>0.0652</td>
<td>0.0526</td>
<td>0.0682</td>
<td><bold>0.0392</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.5068</td>
<td>0.2182</td>
<td>0.2198</td>
<td>0.2278</td>
<td>0.5092</td>
<td>0.2135</td>
<td>0.5788</td>
<td>0.6526</td>
<td>0.5997</td>
<td><bold>0.9342</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.4735</td>
<td>0.4625</td>
<td>0.4523</td>
<td>0.4992</td>
<td>0.4426</td>
<td>0.4019</td>
<td>0.3927</td>
<td>0.3263</td>
<td>0.3893</td>
<td><bold>0.9019</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.2839</td>
<td>0.2019</td>
<td>0.2734</td>
<td>0.2435</td>
<td>0.2283</td>
<td>0.2361</td>
<td>0.2021</td>
<td>0.2172</td>
<td>0.2253</td>
<td><bold>0.1526</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.1353</td>
<td>0.1392</td>
<td>0.1428</td>
<td>0.1028</td>
<td>0.1426</td>
<td>0.1325</td>
<td>0.1397</td>
<td>0.1399</td>
<td>0.1428</td>
<td><bold>0.1028</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.2937</td>
<td>1.1927</td>
<td>1.2739</td>
<td>1.0173</td>
<td>1.4572</td>
<td>1.3920</td>
<td>1.5739</td>
<td>1.4983</td>
<td>1.5028</td>
<td><bold>1.0527</bold></td>
</tr>
<tr>
<td>Enron</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.2058</td>
<td>0.3389</td>
<td>0.3291</td>
<td>0.3009</td>
<td>0.3857</td>
<td>0.3134</td>
<td>0.2945</td>
<td>0.2437</td>
<td>0.2754</td>
<td><bold>0.1971</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.0648</td>
<td>0.0632</td>
<td>0.0592</td>
<td>0.0611</td>
<td>0.0598</td>
<td>0.0534</td>
<td>0.0485</td>
<td>0.0472</td>
<td>0.04173</td>
<td><bold>0.9571</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.4836</td>
<td>0.4982</td>
<td>0.4383</td>
<td>0.4927</td>
<td>0.4892</td>
<td>0.4492</td>
<td>0.4856</td>
<td>0.4139</td>
<td>0.5193</td>
<td><bold>0.9072</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.1233</td>
<td>0.1927</td>
<td>0.1873</td>
<td>0.1865</td>
<td>0.1856</td>
<td>0.1943</td>
<td>0.1827</td>
<td>0.2183</td>
<td>0.1855</td>
<td><bold>0.0859</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.4098</td>
<td>0.9311</td>
<td>0.6921</td>
<td>0.9372</td>
<td>0.7372</td>
<td>0.8922</td>
<td>0.7822</td>
<td>0.8927</td>
<td>0.7212</td>
<td><bold>0.2310</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.9237</td>
<td>1.8729</td>
<td>1.8882</td>
<td>1.9833</td>
<td>1.8727</td>
<td>1.8192</td>
<td>2.1932</td>
<td>2.4821</td>
<td>2.4382</td>
<td><bold>1.3471</bold></td>
</tr>
<tr>
<td>Scene</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.0798</td>
<td>0.0728</td>
<td>0.0732</td>
<td>0.0739</td>
<td>0.0725</td>
<td>0.0732</td>
<td>0.0526</td>
<td>0.0529</td>
<td>0.0592</td>
<td><bold>0.0228</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.7892</td>
<td>0.7256</td>
<td>0.7927</td>
<td>0.8019</td>
<td>0.6299</td>
<td>0.8917</td>
<td>0.6982</td>
<td>0.6625</td>
<td>0.6383</td>
<td><bold>0.9989</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.5638</td>
<td>0.5927</td>
<td>0.5832</td>
<td>0.5725</td>
<td>0.5829</td>
<td>0.5535</td>
<td>0.6093</td>
<td>0.6192</td>
<td>0.6082</td>
<td><bold>0.9282</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.3923</td>
<td>0.3846</td>
<td>0.3626</td>
<td>0.4017</td>
<td>0.3975</td>
<td>0.4103</td>
<td>0.5039</td>
<td>0.5102</td>
<td>0.5097</td>
<td><bold>0.2273</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.1801</td>
<td>0.1578</td>
<td>0.1623</td>
<td>0.1937</td>
<td>0.1643</td>
<td>0.1547</td>
<td>0.2092</td>
<td>0.2176</td>
<td>0.2097</td>
<td><bold>0.1283</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>2.0371</td>
<td>1.9277</td>
<td>1.8235</td>
<td>2.0458</td>
<td>1.8290</td>
<td>1.9366</td>
<td>1.9827</td>
<td>1.8872</td>
<td>1.5279</td>
<td><bold>1.3092</bold></td>
</tr>
<tr>
<td>Yeast</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.1836</td>
<td>0.1936</td>
<td>0.2846</td>
<td>0.1926</td>
<td>0.1872</td>
<td>0.2017</td>
<td>0.2423</td>
<td>0.2978</td>
<td>0.2783</td>
<td><bold>0.1110</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.3760</td>
<td>0.3911</td>
<td>0.4811</td>
<td>0.3812</td>
<td>0.4621</td>
<td>0.4706</td>
<td>0.5312</td>
<td>0.4928</td>
<td>0.4802</td>
<td><bold>0.9019</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.4834</td>
<td>0.3978</td>
<td>0.4378</td>
<td>0.4586</td>
<td>0.3978</td>
<td>0.4837</td>
<td>0.3982</td>
<td>0.3927</td>
<td>0.3826</td>
<td><bold>0.8528</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.1990</td>
<td>0.1830</td>
<td>0.2002</td>
<td>0.1930</td>
<td>0.1932</td>
<td>0.1845</td>
<td>0.2018</td>
<td>0.2418</td>
<td>0.2614</td>
<td><bold>0.0992</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.8921</td>
<td>0.8821</td>
<td>0.8819</td>
<td>0.8913</td>
<td>0.8229</td>
<td>0.7271</td>
<td>0.7897</td>
<td><bold>0.6190</bold></td>
<td>0.8392</td>
<td><bold>0.1590</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.7456</td>
<td>1.9822</td>
<td>1.7393</td>
<td>2.0913</td>
<td>1.7392</td>
<td>2.0381</td>
<td>1.0289</td>
<td>1.2738</td>
<td>1.2473</td>
<td><bold>1.0947</bold></td>
</tr>
<tr>
<td>Slashdot</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.1937</td>
<td>0.2028</td>
<td>0.2947</td>
<td>0.2374</td>
<td>0.2247</td>
<td>0.2750</td>
<td>0.1827</td>
<td>0.1725</td>
<td>0.1889</td>
<td><bold>0.1027</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.3958</td>
<td>0.3358</td>
<td>0.3028</td>
<td>0.3658</td>
<td>0.3579</td>
<td>0.3590</td>
<td>0.4728</td>
<td>0.4526</td>
<td>0.4923</td>
<td><bold>0.8827</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.5992</td>
<td>0.5954</td>
<td>0.5239</td>
<td>0.5539</td>
<td>0.5057</td>
<td>0.5258</td>
<td>0.4928</td>
<td>0.4829</td>
<td>0.4262</td>
<td><bold>0.8927</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.3038</td>
<td>0.4456</td>
<td>0.4937</td>
<td>0.5102</td>
<td>0.5329</td>
<td>0.5683</td>
<td>0.3937</td>
<td>0.3728</td>
<td>0.3576</td>
<td><bold>0.2920</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.1940</td>
<td>0.2249</td>
<td>0.2673</td>
<td>0.2564</td>
<td>0.2847</td>
<td>0.2759</td>
<td>0.1928</td>
<td>0.1829</td>
<td>0.1393</td>
<td><bold>0.1728</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.2478</td>
<td>1.5749</td>
<td>1.5559</td>
<td>1.2854</td>
<td>1.0894</td>
<td>1.9238</td>
<td>1.7283</td>
<td>1.9327</td>
<td>1.3845</td>
<td><bold>1.0489</bold></td>
</tr>
<tr>
<td>Corel5k</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.1024</td>
<td>0.1027</td>
<td>0.1076</td>
<td>0.1085</td>
<td>0.1019</td>
<td>0.1034</td>
<td><bold>0.1005</bold></td>
<td>0.1082</td>
<td>0.1054</td>
<td>0.1007</td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.6737</td>
<td>0.6622</td>
<td>0.6238</td>
<td>0.6492</td>
<td>0.6292</td>
<td>0.6282</td>
<td><bold>0.8918</bold></td>
<td>0.7142</td>
<td>0.6826</td>
<td>0.8354</td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.5529</td>
<td>0.53538</td>
<td>0.5628</td>
<td>0.5281</td>
<td>0.5938</td>
<td>0.5621</td>
<td>0.8836</td>
<td>0.5934</td>
<td>0.6262</td>
<td><bold>0.9382</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.2048</td>
<td>0.2428</td>
<td>0.2738</td>
<td>0.2839</td>
<td>0.2510</td>
<td>0.2472</td>
<td>0.1930</td>
<td>0.2734</td>
<td>0.2438</td>
<td><bold>0.1639</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.1129</td>
<td>0.1927</td>
<td>0.1537</td>
<td>0.1409</td>
<td>0.1728</td>
<td>0.1548</td>
<td>0.1097</td>
<td>0.1382</td>
<td>0.1273</td>
<td><bold>0.1026</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.1289</td>
<td>1.1637</td>
<td>1.1689</td>
<td>1.1927</td>
<td>1.1538</td>
<td>1.1782</td>
<td>1.1849</td>
<td>1.3903</td>
<td>1.1927</td>
<td><bold>1.0937</bold></td>
</tr>
<tr>
<td>Rcv1<break/>(subset 1)</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.1826</td>
<td>0.1973</td>
<td>0.1934</td>
<td>0.1916</td>
<td>0.1423</td>
<td>0.1234</td>
<td>0.1812</td>
<td>0.1221</td>
<td>0.1935</td>
<td><bold>0.0916</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.6934</td>
<td>0.6 812</td>
<td>0.6810</td>
<td>0.6101</td>
<td>0.6091</td>
<td>0.6917</td>
<td>0.6842</td>
<td>0.6817</td>
<td>0.6271</td>
<td><bold>0.9716</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.6937</td>
<td>0.7032</td>
<td>0.6348</td>
<td>0.6634</td>
<td>0.7012</td>
<td>0.6342</td>
<td>0.7845</td>
<td>0.7743</td>
<td>0.7623</td>
<td><bold>0.9238</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.2035</td>
<td>0.2045</td>
<td>0.2859</td>
<td>0.2754</td>
<td>0.2347</td>
<td>0.2854</td>
<td>0.3154</td>
<td>0.2901</td>
<td>0.3017</td>
<td><bold>0.1560</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.2954</td>
<td>0.2018</td>
<td>0.2292</td>
<td>0.3018</td>
<td>0.3045</td>
<td>0.2920</td>
<td>0.3047</td>
<td>0.3081</td>
<td>0.3001</td>
<td><bold>0.1946</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.3929</td>
<td>1.0380</td>
<td>1.5830</td>
<td>1.4924</td>
<td>1.3840</td>
<td>1.3957</td>
<td>2.5498</td>
<td>2.0739</td>
<td>2.3819</td>
<td><bold>1.1471</bold></td>
</tr>
<tr>
<td>Rcv1<break/>(subset 2)</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.0672</td>
<td>0.0628</td>
<td>0.0623</td>
<td>0.0614</td>
<td>0.0693</td>
<td>0.0658</td>
<td><bold>0.0116</bold></td>
<td>0.0159</td>
<td>0.0128</td>
<td>0.0119</td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.5116</td>
<td>0.5232</td>
<td>0.4182</td>
<td>0.5383</td>
<td>0.4019</td>
<td>0.5152</td>
<td>0.7763</td>
<td>0.6528</td>
<td>0.6382</td>
<td><bold>0.8028</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.5373</td>
<td>0.5947</td>
<td>0.5372</td>
<td>0.6027</td>
<td>0.5532</td>
<td>0.5382</td>
<td>0.7928</td>
<td>0.6836</td>
<td>0.6124</td>
<td><bold>0.8352</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.3927</td>
<td>0.3846</td>
<td>0.3253</td>
<td>0.3012</td>
<td>0.3549</td>
<td>0.3328</td>
<td>0.3027</td>
<td>0.4282</td>
<td>0.4153</td>
<td><bold>0.2523</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.1935</td>
<td>0.1823</td>
<td>0.1528</td>
<td>0.1102</td>
<td>0.1538</td>
<td>0.1628</td>
<td>0.0979</td>
<td>0.2093</td>
<td>0.2183</td>
<td><bold>0.0284</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.0928</td>
<td>1.0376</td>
<td>1.0283</td>
<td>1.0192</td>
<td>1.0826</td>
<td>1.0263</td>
<td>1.1028</td>
<td>1.1927</td>
<td>1.1263</td>
<td><bold>1.0378</bold></td>
</tr>
<tr>
<td>tmc2007</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.0927</td>
<td>0.0266</td>
<td>0.0239</td>
<td>0.0264</td>
<td>0.0292</td>
<td>0.0291</td>
<td>0.0527</td>
<td>0.0639</td>
<td>0.0592</td>
<td><bold>0.0128</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.6918</td>
<td>0.7019</td>
<td>0.6827</td>
<td>0.6725</td>
<td>0.6283</td>
<td>0.6629</td>
<td>0.5986</td>
<td>0.5024</td>
<td>0.5756</td>
<td><bold>0.8425</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.6837</td>
<td>0.7038</td>
<td>0.6352</td>
<td>0.6392</td>
<td>0.6643</td>
<td>0.6293</td>
<td>0.6037</td>
<td>0.5833</td>
<td>0.5537</td>
<td><bold>0.8284</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.1916</td>
<td>0.1567</td>
<td>0.2873</td>
<td>0.2273</td>
<td>0.2671</td>
<td>0.1982</td>
<td>0.1283</td>
<td>0.1772</td>
<td>0.1980</td>
<td><bold>0.1330</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.2039</td>
<td>0.1527</td>
<td>0.1826</td>
<td>0.1926</td>
<td>0.1527</td>
<td>0.2081</td>
<td>0.1392</td>
<td>0.1562</td>
<td>0.1982</td>
<td><bold>0.1258</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.9280</td>
<td>1.2327</td>
<td>1.5281</td>
<td>1.3259</td>
<td>1.6299</td>
<td>1.4523</td>
<td>1.0192</td>
<td>1.1728</td>
<td>1.1839</td>
<td><bold>1.0211</bold></td>
</tr>
<tr>
<td>Mediamill</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.1836</td>
<td>0.1936</td>
<td>0.2046</td>
<td>0.1926</td>
<td>0.1872</td>
<td>0.2017</td>
<td>0.1923</td>
<td>0.2078</td>
<td>0.2083</td>
<td><bold>0.1510</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.6076</td>
<td>0.6191</td>
<td>0.6081</td>
<td>0.5 181</td>
<td>0.5062</td>
<td>0.5076</td>
<td>0.6312</td>
<td>0.6928</td>
<td>0.6827</td>
<td><bold>0.8092</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.5834</td>
<td>0.5978</td>
<td>0.5378</td>
<td>0.5586</td>
<td>0.5978</td>
<td>0.5837</td>
<td>0.6982</td>
<td>0.5927</td>
<td>0.5826</td>
<td><bold>0.8528</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.1990</td>
<td>0.1830</td>
<td>0.2002</td>
<td>0.1930</td>
<td>0.1932</td>
<td>0.1845</td>
<td>0.2018</td>
<td>0.2418</td>
<td>0.2614</td>
<td><bold>0.1092</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.3921</td>
<td>0.3821</td>
<td>0.3819</td>
<td>0.3913</td>
<td>0.3229</td>
<td>0.3271</td>
<td>0.3897</td>
<td>0.3190</td>
<td>0.3392</td>
<td><bold>0.2347</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.7456</td>
<td>1.9822</td>
<td>1.7393</td>
<td>2.0913</td>
<td>1.7392</td>
<td>2.0381</td>
<td>1.0289</td>
<td>1.2738</td>
<td>1.2473</td>
<td><bold>1.1947</bold></td>
</tr>
<tr>
<td>Corel16k</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.0203</td>
<td>0.0352</td>
<td>0.0283</td>
<td>0.0384</td>
<td>0.0183</td>
<td>0.0972</td>
<td>0.0255</td>
<td>0.0836</td>
<td>0.0352</td>
<td><bold>0.0152</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.5282</td>
<td>0.5420</td>
<td>0.5920</td>
<td>0.5429</td>
<td>0.5027</td>
<td>0.5829</td>
<td>0.5421</td>
<td>0.59267</td>
<td>0.6826</td>
<td><bold>0.8429</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.5927</td>
<td>0.5018</td>
<td>0.5172</td>
<td>0.5284</td>
<td>0.6038</td>
<td>0.5978</td>
<td>0.5828</td>
<td>0.5527</td>
<td>0.5293</td>
<td><bold>0.9028</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>1.1393</td>
<td>0.1439</td>
<td>0.1595</td>
<td>0.1982</td>
<td>0.2038</td>
<td>0.2573</td>
<td>0.2384</td>
<td>0.2987</td>
<td>0.2674</td>
<td><bold>0.1073</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.1338</td>
<td>0.1872</td>
<td>0.1772</td>
<td>0.1572</td>
<td>0.1027</td>
<td>0.1762</td>
<td>0.1823</td>
<td>0.1563</td>
<td>0.1487</td>
<td><bold>0.0362</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.0013</td>
<td>1.2829</td>
<td>1.1093</td>
<td>1.1256</td>
<td>1.157</td>
<td>1.1982</td>
<td>1.1862</td>
<td>1.1682</td>
<td>1.1579</td>
<td><bold>1.0072</bold></td>
</tr>
<tr>
<td>Genbase</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.0352</td>
<td>0.0283</td>
<td>0.0384</td>
<td>0.0183</td>
<td>0.0572</td>
<td>0.0255</td>
<td>0.0836</td>
<td>0.0352</td>
<td>0.0352</td>
<td><bold>0.0182</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.5027</td>
<td>0.5829</td>
<td>0.4421</td>
<td>0.4926</td>
<td>0.5426</td>
<td>0.5429</td>
<td>0.5082</td>
<td>0.6420</td>
<td>0.5920</td>
<td><bold>0.8429</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.6981</td>
<td>0.6412</td>
<td>0.6725</td>
<td>0.6627</td>
<td>0.6293</td>
<td>0.5928</td>
<td>0.6172</td>
<td>0.6427</td>
<td>0.6912</td>
<td><bold>0.8827</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.3393</td>
<td>0.3039</td>
<td>0.3095</td>
<td>0.3982</td>
<td>0.2938</td>
<td>0.3503</td>
<td>0.3284</td>
<td>0.3987</td>
<td>0.3804</td>
<td><bold>0.1073</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.2038</td>
<td>0.2892</td>
<td>0.2072</td>
<td>0.2592</td>
<td>0.2327</td>
<td>0.2962</td>
<td>0.2923</td>
<td>0.2093</td>
<td>0.2087</td>
<td><bold>0.1062</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.3413</td>
<td>1.3929</td>
<td>1.2903</td>
<td>1.2056</td>
<td>1.2157</td>
<td>1.2982</td>
<td>1.2602</td>
<td>1.1908</td>
<td>1.1987</td>
<td><bold>1.1087</bold></td>
</tr>
<tr>
<td>Bibtex</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.07095</td>
<td>0.07913</td>
<td>0.07252</td>
<td>0.06826</td>
<td>0.07292</td>
<td>0.07162</td>
<td>0.0493</td>
<td>0.0501</td>
<td>0.0498</td>
<td><bold>0.0528</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.7451</td>
<td>0.7519</td>
<td>0.7143</td>
<td>0.7242</td>
<td>0.7092</td>
<td>0.7172</td>
<td>0.7259</td>
<td>0.7583</td>
<td>0.7019</td>
<td><bold>0.8947</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.6091</td>
<td>0.6267</td>
<td>0.6192</td>
<td>0.6815</td>
<td>0.6269</td>
<td>0.6735</td>
<td>0.5628</td>
<td>0.6193</td>
<td>0.5823</td>
<td><bold>0.8628</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.0577</td>
<td>0.0571</td>
<td>0.06921</td>
<td>0.05930</td>
<td>0.06408</td>
<td>0.0598</td>
<td>0.0683</td>
<td>0.06342</td>
<td>0.0610</td>
<td><bold>0.0452</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.01920</td>
<td>0.02043</td>
<td>0.01837</td>
<td>0.02994</td>
<td>0.01937</td>
<td>0.02834</td>
<td>0.01128</td>
<td>0.02893</td>
<td>0.01983</td>
<td><bold>0.0109</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.4937</td>
<td>1.4028</td>
<td>1.5018</td>
<td>1.5937</td>
<td>1.5826</td>
<td>1.4927</td>
<td><bold>1.3272</bold></td>
<td>1.4028</td>
<td>1.3923</td>
<td>1.3321</td>
</tr>
<tr>
<td>Eurlex-sm</td>
<td>Hamming loss <bold>&#x2193;</bold></td>
<td>0.1036</td>
<td>0.1306</td>
<td>0.1146</td>
<td>0.1026</td>
<td>0.1172</td>
<td>0.1097</td>
<td>0.1042</td>
<td>0.1029</td>
<td>0.1027</td>
<td><bold>0.1001</bold></td>
</tr>
<tr>
<td/>
<td>Subset accuracy <bold>&#x2191;</bold></td>
<td>0.5760</td>
<td>0.5911</td>
<td>0.5811</td>
<td>0.5812</td>
<td>0.5621</td>
<td>0.5706</td>
<td>0.5312</td>
<td>0.5928</td>
<td>0.5802</td>
<td><bold>0.8919</bold></td>
</tr>
<tr>
<td/>
<td>Average precision <bold>&#x2191;</bold></td>
<td>0.5834</td>
<td>0.5978</td>
<td>0.5878</td>
<td>0.5486</td>
<td>0.5978</td>
<td>0.5937</td>
<td>0.5702</td>
<td>0.5987</td>
<td>0.5856</td>
<td><bold>0.8528</bold></td>
</tr>
<tr>
<td/>
<td>Ranking loss <bold>&#x2193;</bold></td>
<td>0.1990</td>
<td>0.1830</td>
<td>0.2002</td>
<td>0.1930</td>
<td>0.1932</td>
<td>0.1845</td>
<td>0.2018</td>
<td>0.2418</td>
<td>0.2614</td>
<td><bold>0.0992</bold></td>
</tr>
<tr>
<td/>
<td>One_error <bold>&#x2193;</bold></td>
<td>0.1038</td>
<td>0.1072</td>
<td>0.1172</td>
<td>0.1012</td>
<td>0.1090</td>
<td>0.1062</td>
<td>0.1083</td>
<td>0.1053</td>
<td>0.1047</td>
<td><bold>0.0982</bold></td>
</tr>
<tr>
<td/>
<td>Coverage <bold>&#x2193;</bold></td>
<td>1.8456</td>
<td>1.7822</td>
<td>1.6793</td>
<td>2.2013</td>
<td>1.7992</td>
<td>2.1381</td>
<td>1.7689</td>
<td>1.7838</td>
<td>1.7473</td>
<td><bold>1.1817</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Performance Comparison of Proposed Method against Recent Methods</title>
<p>This paper proposes a novel framework for learning and classifying the imbalanced multi-label data in two phases so that the logistic regression model will work better on imbalanced data. Phase 1 has a pre-processing method named <italic>Borderline MLSMOTE</italic>, which expands minority labels in areas where the concurrent appearance of the minority and majority labels is too high. Phase 2 has an adaptive weighted <italic>l</italic><sub>21</sub>-norm regularized weighted logistic regression to address over-fitting and variable selection. Phase 1 uses data pre-processing to balance imbalanced multi-label data, where Multi-label SMOTE (<italic>MLSMOTE</italic>) is modified to treat labels in the training set as one-<italic>vs.</italic>-rest to create new synthetic samples. Elastic net is modified to the adaptive weighted elastic net to address over-fitting, biased estimation, multi-collinearity, and low false-positive rate. The key challenge to multi-label data lies when the number of labels for prediction is exponential. This involves exploiting label correlation among labels, over-fitting due to high dimensional predictive space, and highly imbalanced training sets. The proposed <italic>Borderline MLSMOTE</italic> is compared with other methods to demonstrate the superiority of the proposed method and is presented in <xref ref-type="table" rid="table-3">Table 3</xref>. The proposed method combines two phases, and phase 2 works on pre-processed data. Most of the other comparison methods only focus on classification, which seems unfair as they work on original datasets.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Comprehensive comparison results between the proposed method and recent algorithms</title>
</caption>
<table frame="hsides">
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th rowspan="2"></th>
<th align="center" colspan="4">Yeast</th>
<th align="center" colspan="4">Enron</th>
<th align="center" colspan="4">Scene</th>
</tr>
<tr>
<th>HL</th>
<th>RL</th>
<th>AP</th>
<th>OE</th>
<th>HL</th>
<th>RL</th>
<th>AP</th>
<th>OE</th>
<th>HL</th>
<th>RL</th>
<th>AP</th>
<th>OE</th>
</tr>
</thead>
<tbody>
<tr>
<td>Shu et al. [<xref ref-type="bibr" rid="ref-26">26</xref>]</td>
<td>0.183</td>
<td>00.158</td>
<td>00.782</td>
<td>0.193</td>
<td>0.161</td>
<td>0.141</td>
<td>0.831</td>
<td>0.257</td>
<td>0.019</td>
<td>0.052</td>
<td>0.791</td>
<td>0.273</td>
</tr>
<tr>
<td>Wu et al. [<xref ref-type="bibr" rid="ref-27">27</xref>]</td>
<td>0.222</td>
<td>00.207</td>
<td>00.709</td>
<td>0.224</td>
<td>0.299</td>
<td>0.278</td>
<td>0.451</td>
<td>0.600</td>
<td>0.021</td>
<td>0.051</td>
<td>0.761</td>
<td>0.314</td>
</tr>
<tr>
<td>Zhang et al. [<xref ref-type="bibr" rid="ref-28">28</xref>]</td>
<td>0.192</td>
<td>00.164</td>
<td>00.771</td>
<td>0.222</td>
<td>0.166</td>
<td>0.149</td>
<td>0.818</td>
<td>0.277</td>
<td>0.036</td>
<td>0.519</td>
<td>0.388</td>
<td>0.972</td>
</tr>
<tr>
<td><bold>Proposed method</bold></td>
<td><bold>0.111</bold></td>
<td><bold>0.099</bold></td>
<td><bold>0.853</bold></td>
<td><bold>0.159</bold></td>
<td><bold>0.152</bold></td>
<td><bold>0.127</bold></td>
<td><bold>0.843</bold></td>
<td><bold>0.197</bold></td>
<td><bold>0.014</bold></td>
<td><bold>0.031</bold></td>
<td><bold>0.921</bold></td>
<td><bold>0.187</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn><p>Note: HL&#x2013;Hamming Loss, RL&#x2013;Ranking Loss, AP&#x2013;Average Precision, OE&#x2013;One Error.</p></fn></table-wrap-foot>
</table-wrap>
<p>In contrast, the proposed method works on new datasets processed by <italic>Borderline MLSMOTE</italic>. Incorporating multiple cluster centers for multi-label learning (IMCC) [38] creates more samples out of neighborhood clustering centers to expand the training set and realize data enhancement. Feature-induced labeling information enrichment for multi-label learning (MLFE) [39] employs the structure information of attribute area to improve label details. Joint Ranking SVM and Binary Relevance with Robust Low-Rank Learning for Multilabel Classification (RBRL) [40] show Ranking SVM and Binary relevance with low-rank solid learning. Three standard data sets have been chosen to validate the legitimacy of the proposed method, yeast (gene function prediction using 2417 samples and 14 labels), image (image classification with 2000 samples and five labels) along with also social (5000 samples with 39 labels). The performance of the proposed framework has been examined with three recent state-of-the-art procedures and confirmed against four metrics. <xref ref-type="table" rid="table-3">Table 3</xref> shows the comparison of the proposed method against different recent works. The proposed method outperforms all metrics in yeast, image, and social dataset. The result shows the proposed method accomplishes a nearly flawless prediction on yeast, image, and social datasets and exhibits the potency of this suggested procedure.</p>

<p>The performance comparison of the proposed system with and without the pre-processing stage is presented in <xref ref-type="fig" rid="fig-6">Figs. 6</xref>&#x2013;<xref ref-type="fig" rid="fig-8">8</xref>. The accuracy increases at reasonable rates for all datasets. Specifically, this framework benefited a lot of data sets with high SCUBMLE values. The datasets with high concurrent value are Corel5k, Mediamill, CAL500, Enron, Corel6k, and both Rcv1 data benefitted greatly from this framework. The accuracy of CAL500 increases to 31%, and the accuracy is 92%. With Emotions and Mediamill the percentage increase is 27.53% and 19.96%. The accuracy of Emotions is 87.43% and 80.92%. The overall increase in accuracy ranges from 7% to 31%. The overall increase in Precision for all datasets ranges from 3 % to 53%. The Corel5k dataset with high concurrence measures benefits greatly from this framework in terms of Precision&#x2014;the precision value for the Corel5k dataset increases from 40% to 93%. Mediamill showed 50%; with the proposed framework, it gives 85% precision. The Scene dataset offers 87% without phase 1, which now provides 92% precision. The increase in the percentage of Precision after the framework is as follows: CAL500 (28%), Enron (37.36%), Rcv1 (Subset2) (25.14%), Eurlex-sm (24.44%), Corel6k (23.01%), tmc2007 (13.77%).</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Hamming loss performance of adaptive weighted elastic net with and without pre-processing</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34373-fig-6.tif"/>
</fig><fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Accuracy of adaptive weighted elastic net with and without pre-processing</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34373-fig-7.tif"/>
</fig><fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Performance of adaptive weighted elastic net with and without pre-processing</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34373-fig-8.tif"/>
</fig>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusion</title>
<p>A framework to classify and predict imbalanced multi-label data has been introduced in this work. The framework has two phases; (1) an adaptive <italic>Borderline&#x2013;MLSMOTE</italic> has proposed and pre-processed the biased data, and (2) <italic>l</italic><sub>21</sub>-norm (Elastic net) regularized adaptive weighted logistic regression has exploited to learn parameters and predict the processed data. This variant of <italic>MLSMOTE</italic> concentrates on minority concurrence labels that contribute to the relief imbalance among multiple labels and promote the influence of minority labels. Experimental effects on various multi-label datasets have shown that the proposed framework enhances the performance over other competing and recent methods in most cases. The results confirm that the dataset with concurrent high labels benefited greatly from the proposed system. The proposed <italic>Borderline&#x2013;MLSMOTE</italic> method works based on kNN to generate new samples. Identification of the k value may be challenging in <italic>Borderline&#x2013;MLSMOTE</italic>. Further investigations on the imbalance of hierarchical data can be done on <italic>Borderline&#x2013;MLSMOTE</italic>. The proposed sampling method is poor in identifying label correlations; additional work on the above issue could throw light on multi-label data.</p>
</sec>
</body>
<back>
<ack>
<p>This research was partly supported by the Technology Development Program of MSS (No. S3033853) and by the National Research Foundation of Korea (NRF) grant funded by the Korea government (MSIT) (No. 2021R1A4A1031509).</p>
</ack>
<sec><title>Funding Statement</title>
<p>The authors received no specific funding for this study.</p>
</sec>
<sec><title>Author Contributions</title>
<p>P. K. A. Chitra and S. Geetha contributed in conceptualization and design of the study and supervising the research process, S. Appavu alias Balamurugan and S. Geetha drafted the introduction and discussion sections, S. Geetha, Seifedine Kadry and Jungeun Kim conducted literature review and sourced relevant studies, P. K. A. Chitra and S. Geetha involved in writing the methodology section and developing research instruments, Jungeun Kim and Keejun Han conducted experiments and field work, P. K. A. Chitra and S. Geetha involved in statistical analysis and interpretation of results, P. K. A. Chitr, S. Geetha and Seifedine Kadry reviewed and edited final manuscript for submission. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability"><title>Availability of Data and Materials</title>
<p>The data used in this research are open source and can be downloaded from <ext-link ext-link-type="uri" xlink:href="https://mulan.sourceforge.net/datasets-mlc.html">https://mulan.sourceforge.net/datasets-mlc.html</ext-link> (accessed on 15 September 2022).</p>
</sec>
<sec sec-type="COI-statement"><title>Conflicts of Interest</title>
<p>The authors declare that they have no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. L.</given-names> <surname>Zhang</surname></string-name> and <string-name><given-names>Z. H.</given-names> <surname>Zhou</surname></string-name></person-group>, &#x201C;<article-title>ML-KNN: A lazy learning approach to multi-label learning</article-title>,&#x201D; <source>Pattern Recognit.</source>, vol. <volume>40</volume>, no. <issue>7</issue>, pp. <fpage>2038</fpage>&#x2013;<lpage>2048</lpage>, <year>2007</year>. doi: <pub-id pub-id-type="doi">10.1016/j.patcog.2006.12.019</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Feng</surname></string-name> and <string-name><given-names>D.</given-names> <surname>Xu</surname></string-name></person-group>, &#x201C;<article-title>Transductive multi-instance multi-label learning algorithm with application to automaticimage annotation</article-title>,&#x201D; <source>Expert. Syst. Appl.</source>, vol. <volume>37</volume>, no. <issue>1</issue>, pp. <fpage>661</fpage>&#x2013;<lpage>670</lpage>, <year>2010</year>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y. C.</given-names> <surname>Chang</surname></string-name>, <string-name><given-names>S. M.</given-names> <surname>Chen</surname></string-name>, and <string-name><given-names>C. J.</given-names> <surname>Liau</surname></string-name></person-group>, &#x201C;<article-title>Multilabel text categorization based on a new linear classifier learning method and a category-sensitive refinement method</article-title>,&#x201D; <source>Expert Syst. Appl.</source>, vol. <volume>34</volume>, no. <issue>3</issue>, pp. <fpage>1948</fpage>&#x2013;<lpage>1953</lpage>, <year>2008</year>. doi: <pub-id pub-id-type="doi">10.1016/j.eswa.2007.02.037</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>R. M. M.</given-names> <surname>Vallim</surname></string-name>, <string-name><given-names>T. S.</given-names> <surname>Duque</surname></string-name>, <string-name><given-names>D. E.</given-names> <surname>Goldberg</surname></string-name>, and <string-name><given-names>A. C.</given-names> <surname>Carvalho</surname></string-name></person-group>, &#x201C;<article-title>The multi-label OCS with a genetic algorithm for rule discovery: Implementation and first results</article-title>,&#x201D; in <conf-name>Proc. 11th Annu. Conf. Genetic Evol. Comput.</conf-name>, <year>2009</year>, pp. <fpage>1323</fpage>&#x2013;<lpage>1330</lpage>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Trohidis</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Tsoumakas</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Kalliris</surname></string-name>, and <string-name><given-names>I. P.</given-names> <surname>Vlahavas</surname></string-name></person-group>, &#x201C;<article-title>Multi-label classification of music into emotions</article-title>,&#x201D; in <conf-name>Proc. 9th Int. Conf. Music Inform. Retr.</conf-name>, <year>2008</year>, vol. <volume>8</volume>, pp. <fpage>325</fpage>&#x2013;<lpage>330</lpage>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Luoand</surname></string-name>, and <string-name><given-names>J.</given-names> <surname>Zhang</surname></string-name></person-group>, &#x201C;<article-title>Predicting drug side effects by multi-label learning and ensemble learning</article-title>,&#x201D; <source>BMC Bioinform.</source>, vol. <volume>16</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>11</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>He</surname></string-name> and <string-name><given-names>E. A.</given-names> <surname>Garcia</surname></string-name></person-group>, &#x201C;<article-title>Learning from imbalanced data</article-title>,&#x201D; <source>IEEE Trans. Knowl. Data Eng.</source>, vol. <volume>21</volume>, no. <issue>9</issue>, pp. <fpage>1263</fpage>&#x2013;<lpage>1284</lpage>, <year>2009</year>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>N.</given-names> <surname>Japkowicz</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Stephen</surname></string-name></person-group>, &#x201C;<article-title>The class imbalance problem: A systematic study</article-title>,&#x201D; <source>Intell. Data Anal.</source>, vol. <volume>6</volume>, no. <issue>5</issue>, pp. <fpage>429</fpage>&#x2013;<lpage>449</lpage>, <year>2002</year>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Batuwita</surname></string-name> and <string-name><given-names>V.</given-names> <surname>Palade</surname></string-name></person-group>, &#x201C;<article-title>Class imbalance learning methods for support vector machines</article-title>,&#x201D; <source>Imbalanced Learn.: Found., Algorithms, Appl.</source>, vol. <volume>20</volume>, no. <issue>3</issue>, pp. <fpage>83</fpage>&#x2013;<lpage>99</lpage>, <year>2013</year>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J. H.</given-names> <surname>Xue</surname></string-name> and <string-name><given-names>P.</given-names> <surname>Hall</surname></string-name></person-group>, &#x201C;<article-title>Why does rebalancing class-unbalanced data improve AUC for linear discriminant analysis?</article-title>,&#x201D; <source>IEEE Trans. Pattern Anal. Mach. Intell.</source>, vol. <volume>37</volume>, no. <issue>5</issue>, pp. <fpage>1109</fpage>&#x2013;<lpage>1112</lpage>, <year>2014</year>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. L.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>Y. K.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Yang</surname></string-name>, and <string-name><given-names>X. Y.</given-names> <surname>Liu</surname></string-name></person-group>, &#x201C;<article-title>Towards class-imbalance aware multi-label learning</article-title>,&#x201D; <source>IEEE Trans. Cybern.</source>, vol. <volume>52</volume>, no. <issue>6</issue>, pp. <fpage>4459</fpage>&#x2013;<lpage>4471</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>N. V.</given-names> <surname>Chawla</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Sylvester</surname></string-name></person-group>, &#x201C;<article-title>Exploiting diversity in ensembles: Improving the performance on unbalanced datasets</article-title>,&#x201D; in <conf-name>Int. Workshop Multiple Classif. Syst.</conf-name>, <year>2007</year>, pp. <fpage>397</fpage>&#x2013;<lpage>406</lpage>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Han</surname></string-name>, <string-name><given-names>W. Y.</given-names> <surname>Wang</surname></string-name>, and <string-name><given-names>B. H.</given-names> <surname>Mao</surname></string-name></person-group>, &#x201C;<article-title>Borderline-SMOTE: A new over-sampling method in imbalanced data sets learning</article-title>,&#x201D; in <conf-name> Int. Conf. Intell. Comput.</conf-name>, <year>2005</year>, pp. <fpage>878</fpage>&#x2013;<lpage>887</lpage>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Q.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Luo</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Huang</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Feng</surname></string-name>, and <string-name><given-names>Z.</given-names> <surname>Liu</surname></string-name></person-group>, &#x201C;<article-title>Novel ensemble method for imbalanced data learning: Bagging of extrapolation-SMOTE SVM</article-title>,&#x201D; <source>Comput. Intell. Neurosci.</source>, vol. <volume>2017</volume>, no. <issue>3</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>11</lpage>, <year>2017</year>. doi: <pub-id pub-id-type="doi">10.1155/2017/1827016</pub-id>; <pub-id pub-id-type="pmid">28250765</pub-id></mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>R. C.</given-names> <surname>Bhagat</surname></string-name> and <string-name><given-names>S. S.</given-names> <surname>Patil</surname></string-name></person-group>, &#x201C;<article-title>Enhanced SMOTE algorithm for classification of imbalanced big-data using random forest</article-title>,&#x201D; in <conf-name>2015 IEEE Int. Adv. Comput. Conf. (IACC)</conf-name>, <publisher-loc>Banglore, India</publisher-loc>, <year>2015</year>, pp. <fpage>403</fpage>&#x2013;<lpage>408</lpage>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Q.</given-names> <surname>Gu</surname></string-name>, <string-name><given-names>X. M.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Wu</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Ningand</surname></string-name>, and <string-name><given-names>C. S.</given-names> <surname>Xin</surname></string-name></person-group>, &#x201C;<article-title>An improved SMOTE algorithm based on genetic algorithm for imbalanced data classification</article-title>,&#x201D; <source>J. Digit. Inform. Manage.</source>, vol. <volume>14</volume>, no. <issue>2</issue>, pp. <fpage>92</fpage>&#x2013;<lpage>103</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Charte</surname></string-name>, <string-name><given-names>A. J.</given-names> <surname>Rivera</surname></string-name>, <string-name><given-names>M. J.</given-names> <surname>delJesus</surname></string-name>, and <string-name><given-names>F.</given-names> <surname>Herrera</surname></string-name></person-group>, &#x201C;<article-title>MLSMOTE: Approaching imbalanced multi-label learning through synthetic instance generation</article-title>,&#x201D; <source>Knowl. Based Syst.</source>, vol. <volume>89</volume>, no. <issue>1</issue>, pp. <fpage>385</fpage>&#x2013;<lpage>397</lpage>, <year>2015</year>. doi: <pub-id pub-id-type="doi">10.1016/j.knosys.2015.07.019</pub-id>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Popli</surname></string-name>, <string-name><given-names>I.</given-names> <surname>Kansal</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Garg</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Goyal</surname></string-name>, and <string-name><given-names>K.</given-names> <surname>Garg</surname></string-name></person-group>, &#x201C;<article-title>Classification and recognition of online handwritten alphabets using machine learning methods</article-title>,&#x201D; <conf-name> IOP Conf. Series: Mat. Sci. Eng.</conf-name>, <year>2021</year>, vol. <volume>1022</volume>, <issue>Art. no. 012111</issue>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A. M.</given-names> <surname>Mishra</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>A deep learning based novel approach for weed growth estimation</article-title>,&#x201D; <source>Intell. Autom. Soft Comput.</source>, vol. <volume>31</volume>, no. <issue>2</issue>, pp. <fpage>1157</fpage>&#x2013;<lpage>1172</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Sharma</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Mittal</surname></string-name>, and <string-name><given-names>N.</given-names> <surname>Goyal</surname></string-name></person-group>, &#x201C;<article-title>An assessment of machine learning and deep learning techniques with applications</article-title>,&#x201D; <source>ECS Trans.</source>, vol. <volume>107</volume>, no. <issue>1</issue>, pp. <fpage>8979</fpage>&#x2013;<lpage>8988</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>V.</given-names> <surname>Verma</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>A deep learning based intelligent garbage detection system using an unmanned aerial vehicle</article-title>,&#x201D; <source>Symmetry</source>, vol. <volume>14</volume>, no. <issue>5</issue>, <year>2022</year>, <comment>Art. no. 960</comment>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Charte</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Rivera</surname></string-name>, <string-name><given-names>M. J. D.</given-names> <surname>Jesus</surname></string-name>, and <string-name><given-names>F.</given-names> <surname>Herrera</surname></string-name></person-group>, &#x201C;<article-title>A first approach to deal with imbalance in multi-label datasets</article-title>,&#x201D; in <conf-name>Int. Conf. Hybrid Artif. Intell. Syst.</conf-name>, <year>2013</year>, pp. <fpage>150</fpage>&#x2013;<lpage>160</lpage>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Napierala</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Stefanowski</surname></string-name></person-group>, &#x201C;<article-title>Types of minority class examples and their influence on learning classifiers from imbalanced data</article-title>,&#x201D; <source>J. Intell. Inform. Syst.</source>, vol. <volume>46</volume>, no. <issue>3</issue>, pp. <fpage>563</fpage>&#x2013;<lpage>597</lpage>, <year>2016</year>. doi: <pub-id pub-id-type="doi">10.1007/s10844-015-0368-1</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Zou</surname></string-name> and <string-name><given-names>H. H.</given-names> <surname>Zhang</surname></string-name></person-group>, &#x201C;<article-title>On the adaptive elastic-net with a diverging number of parameters</article-title>,&#x201D; <source>Ann. Stat.</source>, vol. <volume>37</volume>, no. <issue>4</issue>, pp. <fpage>1733</fpage>&#x2013;<lpage>1751</lpage>, <year>2009</year>; <pub-id pub-id-type="pmid">20445770</pub-id></mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Liu</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Zhang</surname></string-name></person-group>, &#x201C;<article-title>MLSLR: Multilabel learning via sparse logistic regression</article-title>,&#x201D; <source>Inf. Sci.</source>, vol. <volume>281</volume>, no. <issue>3</issue>, pp. <fpage>310</fpage>&#x2013;<lpage>320</lpage>, <year>2014</year>. doi: <pub-id pub-id-type="doi">10.1016/j.ins.2014.05.013</pub-id>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Shu</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Lv</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Yan</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>S.</given-names> <surname>He</surname></string-name>, and <string-name><given-names>J.</given-names> <surname>He</surname></string-name></person-group>, &#x201C;<article-title>Incorporating multiple cluster centers for multi-label learning</article-title>,&#x201D; <source>Inf. Sci.</source>, vol. <volume>590</volume>, no. <issue>8</issue>, pp. <fpage>60</fpage>&#x2013;<lpage>73</lpage>, <year>2022</year>. doi: <pub-id pub-id-type="doi">10.1016/j.ins.2021.12.104</pub-id>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Wu</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Zheng</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Tian</surname></string-name>, and <string-name><given-names>D.</given-names> <surname>Liu</surname></string-name></person-group>, &#x201C;<article-title>Joint ranking SVM and binary relevance with robust low-rank learningfor multi-label classification</article-title>,&#x201D; <source>Neural Netw.</source>, vol. <volume>122</volume>, no. <issue>3</issue>, pp. <fpage>24</fpage>&#x2013;<lpage>39</lpage>, <year>2020</year>; <pub-id pub-id-type="pmid">31675625</pub-id></mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>Q. W.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Zhong</surname></string-name>, and <string-name><given-names>M. L.</given-names> <surname>Zhang</surname></string-name></person-group>, &#x201C;<article-title>Feature-induced labeling information enrichment for multi-label learning</article-title>,&#x201D; in <conf-name>Proc. AAAI Conf. Artif. Intell.</conf-name>, <year>2018</year>, vol. <volume>32</volume>.</mixed-citation></ref>
</ref-list>
</back></article>