<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">68044</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2025.068044</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Federated Multi-Label Feature Selection via Dual-Layer Hybrid Breeding Cooperative Particle Swarm Optimization with Manifold and Sparsity Regularization</article-title>
<alt-title alt-title-type="left-running-head">Federated Multi-Label Feature Selection via Dual-Layer Hybrid Breeding Cooperative Particle Swarm Optimization with Manifold and Sparsity Regularization</alt-title>
<alt-title alt-title-type="right-running-head">Federated Multi-Label Feature Selection via Dual-Layer Hybrid Breeding Cooperative Particle Swarm Optimization with Manifold and Sparsity Regularization</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Zhang</surname><given-names>Songsong</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Jin</surname><given-names>Huazhong</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref><email>jinhuazhong@hbut.edu.cn</email></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Ye</surname><given-names>Zhiwei</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Yang</surname><given-names>Jia</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Zhang</surname><given-names>Jixin</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Wu</surname><given-names>Dongfang</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Zheng</surname><given-names>Xiao</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western"><surname>Song</surname><given-names>Dingfeng</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<aff id="aff-1"><label>1</label><institution>School of Computer Science, Hubei University of Technology</institution>, <addr-line>Wuhan, 430000</addr-line>, <country>China</country></aff>
<aff id="aff-2"><label>2</label><institution>Hubei Provincial Key Laboratory of Green Intelligent Computing Power Network</institution>, <addr-line>Wuhan, 430000</addr-line>, <country>China</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Huazhong Jin. Email: <email>jinhuazhong@hbut.edu.cn</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2025</year></pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>10</day>
<month>11</month>
<year>2025</year>
</pub-date>
<volume>86</volume>
<issue>1</issue>
<fpage>1</fpage>
<lpage>19</lpage>
<history>
<date date-type="received">
<day>20</day>
<month>5</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>8</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 The Authors.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_68044.pdf"></self-uri>
<abstract>
<p>Multi-label feature selection (MFS) is a crucial dimensionality reduction technique aimed at identifying informative features associated with multiple labels. However, traditional centralized methods face significant challenges in privacy-sensitive and distributed settings, often neglecting label dependencies and suffering from low computational efficiency. To address these issues, we introduce a novel framework, Fed-MFSDHBCPSO&#x2014;federated MFS via dual-layer hybrid breeding cooperative particle swarm optimization algorithm with manifold and sparsity regularization (DHBCPSO-MSR). Leveraging the federated learning paradigm, Fed-MFSDHBCPSO allows clients to perform local feature selection (FS) using DHBCPSO-MSR. Locally selected feature subsets are encrypted with differential privacy (DP) and transmitted to a central server, where they are securely aggregated and refined through secure multi-party computation (SMPC) until global convergence is achieved. Within each client, DHBCPSO-MSR employs a dual-layer FS strategy. The inner layer constructs sample and label similarity graphs, generates Laplacian matrices to capture the manifold structure between samples and labels, and applies <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>-norm regularization to sparsify the feature subset, yielding an optimized feature weight matrix. The outer layer uses a hybrid breeding cooperative particle swarm optimization algorithm to further refine the feature weight matrix and identify the optimal feature subset. The updated weight matrix is then fed back to the inner layer for further optimization. Comprehensive experiments on multiple real-world multi-label datasets demonstrate that Fed-MFSDHBCPSO consistently outperforms both centralized and federated baseline methods across several key evaluation metrics.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Multi-label feature selection</kwd>
<kwd>federated learning</kwd>
<kwd>manifold regularization</kwd>
<kwd>sparse constraints</kwd>
<kwd>hybrid breeding optimization algorithm</kwd>
<kwd>particle swarm optimizatio algorithm</kwd>
<kwd>privacy protection</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>National Natural Science Foundation of China</funding-source>
<award-id>U23A20318</award-id>
<award-id>62376089</award-id>
<award-id>62302153</award-id>
<award-id>62302154</award-id>
</award-group>
<award-group id="awg2">
<funding-source>Key Research and Development Program of Hubei Province</funding-source>
<award-id>2023BEB024</award-id>
</award-group>
<award-group id="awg3">
<funding-source>Young and Middle-Aged Researchers in Hubei Higher Education Institutions</funding-source>
<award-id>T2023007</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>In recent years, rapid advancements in network and communication technologies have facilitated the integration of intelligent transportation systems, smart healthcare, and the Internet of Things (IoT), generating vast amounts of high-dimensional multi-label data. Applications such as image annotation, text classification, and personalized recommendation increasingly rely on multi-label learning, where instances can be associated with multiple labels&#x2014;distinct from traditional single-label classification [<xref ref-type="bibr" rid="ref-1">1</xref>]. However, as data dimensionality grows, the presence of redundant and noisy features leads to the &#x201C;curse of dimensionality,&#x201D; which undermines training efficiency, heightens the risk of overfitting, and degrades generalization performance [<xref ref-type="bibr" rid="ref-2">2</xref>]. Consequently, effective feature selection (FS) in multi-label contexts has become a critical challenge.</p>
<p>With the rise of IoT, traditional centralized data processing faces issues such as high latency and privacy risks. To address the need for low-latency, localized processing, many IoT applications are adopting distributed architectures like edge computing [<xref ref-type="bibr" rid="ref-3">3</xref>] and Federated Learning (FL) [<xref ref-type="bibr" rid="ref-4">4</xref>]. Yet, federated settings pose significant problems for FS algorithms due to heterogeneous data distributions that are non-independent and identically distributed (non-IID), limited communication bandwidth, and strict privacy requirements. This calls for algorithms that not only preserve privacy and improve computational efficiency but also ensure effective and reliable FS.</p>
<p>In multi-label learning, the presence of strong inter-label dependencies significantly increases the complexity of FS. Existing methods typically employ first-order (independent labels), second-order (pairwise label relationships), or higher-order (complex dependency structures) strategies for FS [<xref ref-type="bibr" rid="ref-5">5</xref>], often augmented by advanced techniques such as Graph Neural Networks (GNNs) [<xref ref-type="bibr" rid="ref-6">6</xref>], manifold learning [<xref ref-type="bibr" rid="ref-7">7</xref>], semi-supervised approaches [<xref ref-type="bibr" rid="ref-8">8</xref>], long short-term memory (LSTM) [<xref ref-type="bibr" rid="ref-9">9</xref>], and incremental learning [<xref ref-type="bibr" rid="ref-10">10</xref>] to enhance performance. Despite notable advancements, these methods continue to face considerable challenges in high-dimensional, non-IID data scenarios, including susceptibility to the local optima, low solution quality in multi-label feature selection (MFS), and limited generalizability.</p>
<p>To overcome these issues, researchers have explored meta-heuristic algorithms (MAs) such as Genetic Algorithms (GA) [<xref ref-type="bibr" rid="ref-11">11</xref>], Ant Colony Optimization algorithm (ACO) [<xref ref-type="bibr" rid="ref-12">12</xref>], and Particle Swarm Optimization algorithm (PSO) [<xref ref-type="bibr" rid="ref-13">13</xref>], which enhance global search capabilities and improve the stability of FS. Ye et al. [<xref ref-type="bibr" rid="ref-14">14</xref>] proposed the Hybrid Breeding Optimization algorithm (HBO), inspired by the heterosis theory of Chinese hybrid rice breeding. The algorithm partitions rice into three groups&#x2014;maintainer, sterile, and restorers&#x2014;based on fitness ranking, and these groups evolve cooperatively to produce superior offspring. This method has demonstrated strong performance in FS [<xref ref-type="bibr" rid="ref-13">13</xref>] and MFS tasks [<xref ref-type="bibr" rid="ref-15">15</xref>]. PSO, a well-established heuristic algorithm, is recognized for its robustness and rapid convergence, excelling in global search but often susceptible to being trapped in the local optima. Wang et al. [<xref ref-type="bibr" rid="ref-16">16</xref>] proposed a distance-based multi-objective PSO, which fully leverages PSO&#x2019;s global search capability, incorporating adaptive distance and position update strategies to enhance optimization performance. Zhong et al. [<xref ref-type="bibr" rid="ref-17">17</xref>] introduced the self-adaptive competitive swarm optimizer, which exploits PSO&#x2019;s global search advantage and combines parameter sorting and population reduction strategies to achieve superior performance across multiple benchmark and real-world problems. In contrast, the co-evolutionary mechanism of HBO refines the search trajectory, alleviating premature convergence and enhancing stability. The integration of these two algorithms provides a promising solution to enhance both the performance and stability of MFS models.</p>
<p>While recent efforts to integrate FL with MFS mainly utilize methods such as mutual information [<xref ref-type="bibr" rid="ref-18">18</xref>], fuzzy information theory [<xref ref-type="bibr" rid="ref-19">19</xref>], or causal inference [<xref ref-type="bibr" rid="ref-20">20</xref>], these approaches often fall short of fully capturing complex feature interactions and rely solely on the inherent data aggregation of FL for privacy protection. This is insufficient to safeguard sensitive information during large-scale transmission, leaving the system vulnerable to privacy breaches. Furthermore, most methods overlook local manifold structures and redundancy in high-dimensional data, leading to unstable performance under non-IID conditions.</p>
<p>To address these challenges, we propose Fed-MFSDHBCPSO&#x2014;a novel framework for federated MFS, based on dual-layer hybrid breeding cooperative PSO with manifold and sparsity regularization (DHBCPSO-MSR). Built upon a FL architecture, Fed-MFSDHBCPSO integrates manifold regularization, <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>-norm sparsity constraints, and a hybrid breeding cooperative PSO (HBCPSO). To ensure rigorous privacy protection, it further incorporates differential privacy (DP) [<xref ref-type="bibr" rid="ref-21">21</xref>] and secure multi-party computation (SMPC) [<xref ref-type="bibr" rid="ref-22">22</xref>]. In this framework, each client locally performs dual-layer FS using DHBCPSO-MSR to produce an optimal feature subset, which is then encrypted and sent to the server. The server aggregates the encrypted results using SMPC, updates the global feature set, and redistributes it to clients for iterative refinement until convergence. The HBCPSO combines the triple-population co-evolution strategy of HBO with the global search capabilities of PSO, enhancing both stability and search efficiency. Extensive experiments on multiple real-world multi-label datasets demonstrate that Fed-MFSDHBCPSO consistently outperforms existing centralized and federated FS methods in terms of average precision, coverage, macro-F1, micro-F1, hamming loss, and ranking loss, with particularly significant improvements under non-IID conditions. The main contributions of this paper are as follows.
<list list-type="bullet">
<list-item>
<p>We propose Fed-MFSDHBCPSO, a FL framework that enables privacy-preserving and efficient distributed MFS by executing DHBCPSO-MSR locally on each client, while integrating DP and SMPC to ensure encrypted transmission and secure aggregation.</p></list-item>
<list-item>
<p>We propose DHBCPSO-MSR, a dual-layer MFS algorithm. The inner layer optimizes the FS weight matrix through manifold regularization and sparsity constraints on both samples and labels, enhancing feature discriminability. The outer layer further refines the weight matrix obtained from the inner layer using HBCPSO, ultimately achieving the optimal FS weight matrix.</p></list-item>
<list-item>
<p>We propose HBCPSO, an algorithm inspired by the co-evolutionary mechanism of hybrid breeding optimization (HBO) based on heterosis theory. The population is divided into maintainer, sterile, and restorer lines based on fitness ranking, with each line optimized through co-evolution. PSO is applied within each line for global search. The optimal solution information is exchanged periodically among populations, resulting the generation of superior offspring. HBCPSO effectively avoids local sparsity traps and improves the global optimization of FS.</p></list-item>
</list></p>
<p>The remainder of this paper is structured as follows. <xref ref-type="sec" rid="s2">Section 2</xref> reviews the related work, and <xref ref-type="sec" rid="s3">Section 3</xref> presents the proposed method in detail. <xref ref-type="sec" rid="s4">Section 4</xref> reports experimental results on real-world datasets to validate the effectiveness of the approach. Finally, <xref ref-type="sec" rid="s5">Section 5</xref> concludes the paper.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Work</title>
<p>Most early studies have focused on centralized MFS, while research on federated approaches for multi-label datasets remains limited. To date, only a few works have explored federated MFS. The following section reviews these studies.</p>
<sec id="s2_1">
<label>2.1</label>
<title>Centralized Multi-Label Feature Selection</title>
<p>MFS methods are typically classified into two main categories: problem transformation and algorithm adaptation. Problem transformation approaches convert multi-label tasks into multiple single-label problems, enabling the application of traditional single-label FS techniques such as entropy-based label assignment (ELA), binary relevance (BR), and label powerset (LP) [<xref ref-type="bibr" rid="ref-23">23</xref>]. However, BR neglects the dependencies among labels, whereas LP often suffers from class imbalance and increased computational complexity.</p>
<p>Algorithm adaptation techniques overcome these limitations by directly extending single-label FS methods to handle multi-label data. Representative strategies include mutual information-based approaches for evaluating feature relevance and redundancy, as well as causal discovery frameworks for identifying causally informative features [<xref ref-type="bibr" rid="ref-24">24</xref>]. Moreover, MAs such as GA [<xref ref-type="bibr" rid="ref-11">11</xref>], ACO [<xref ref-type="bibr" rid="ref-12">12</xref>] and PSO [<xref ref-type="bibr" rid="ref-13">13</xref>] have been successfully employed to explore high-dimensional feature spaces.</p>
<p>Sparse learning has become a key focus in recent MFS research [<xref ref-type="bibr" rid="ref-25">25</xref>], with <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:msub><mml:mi>L</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:math></inline-formula>-norm and <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>-norm regularizations as mainstream techniques. Huang et al. [<xref ref-type="bibr" rid="ref-26">26</xref>] proposed LLSF, using <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:msub><mml:mi>L</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:math></inline-formula>-norm regularization to learn low-dimensional representations while preserving label dependencies. Jian et al. [<xref ref-type="bibr" rid="ref-27">27</xref>] combined <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>-norm regularization with latent space mapping to improve the performance of FS. Manifold learning methods, such as the PMFS-LRS based on low-rank sparse factorization proposed by Sun et al. [<xref ref-type="bibr" rid="ref-28">28</xref>], decompose the candidate label matrix into low-rank and sparse components, enabling the distinction between ground-truth and noisy labels, thus addressing performance degradation in traditional methods. Additionally, Zhang et al. [<xref ref-type="bibr" rid="ref-29">29</xref>] introduced manifold regularization to model label correlations and enhance feature relevance, while Zhang et al. [<xref ref-type="bibr" rid="ref-30">30</xref>] proposed LRDG, an MFS method that optimizes feature selection through deep learning-based latent representation and pseudo-label learning.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Federated Multi-Label Feature Selection</title>
<p>Federated FS has recently advanced in both single-label and multi-label domains [<xref ref-type="bibr" rid="ref-31">31</xref>]. Inspired by the FL paradigm, existing methods are generally categorized into vertical and horizontal federated FS. Vertical FS applies when clients share the same instance IDs but possess different feature spaces [<xref ref-type="bibr" rid="ref-32">32</xref>], whereas horizontal FS is used when clients hold different data instances while sharing a common feature set [<xref ref-type="bibr" rid="ref-33">33</xref>].</p>
<p>Within horizontal federated MFS, Mahanipour and Khamfroush [<xref ref-type="bibr" rid="ref-18">18</xref>] introduced FMLFS, which employs information-theoretic measures to evaluate feature-label associations and mitigate redundancy. They further proposed a fuzzy logic-based approach that integrates reinforcement learning with ACO [<xref ref-type="bibr" rid="ref-19">19</xref>]. Similarly, Song et al. [<xref ref-type="bibr" rid="ref-20">20</xref>] developed FedCMFS, a causal federated MFS method incorporating three novel components to enhance the performance of FS. However, most existing methods rely solely on the inherent data-locality of FL for privacy protection, without incorporating more rigorous mechanisms such as DP or SMPC&#x2014;leaving them susceptible to potential information leakage.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>The Proposed Method</title>
<p>In this section, we provide a detailed introduction to Fed-MFSHBCPSO and the locally deployed DHBCPSO-MSR algorithm within the federated environment, along with an analysis of their privacy preservation, communication, and computational complexities. To ensure clarity and accuracy in presenting these details, in the preparation of this paper, AI tools (e.g., ChatGPT 4.1 and ChatGPT 5) are utilized for language polishing and structural adjustments.</p>
<sec id="s3_1">
<label>3.1</label>
<title>A Dual-Layer Hybrid Breeding Cooperative Particle Swarm Optimization Algorithm with Manifold and Sparsity Regularization (DHBCPSO-MSR)</title>
<p>This subsection introduces the DHBCPSO-MSR algorithm, which is executed by the client in the proposed Fed-MFSHBCPSO framework, including its inner modeling layer and outer optimization layer. The detailed flowchart of DHBCPSO-MSR is shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>The flowchart of DHBCPSO-MSR</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-1.tif"/>
</fig>
<sec id="s3_1_1">
<label>3.1.1</label>
<title>Inner Modeling Layer of DHBCPSO-MSR</title>
<p><bold>Step 1: Initialization of input data.</bold> We Consider a multi-label dataset where <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mi>X</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is the feature matrix with <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mi>n</mml:mi></mml:math></inline-formula> samples and <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:mi>d</mml:mi></mml:math></inline-formula> features, and <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mi>Y</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>q</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is the label matrix with <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:mi>q</mml:mi></mml:math></inline-formula> labels. Each row of <italic>X</italic> and <italic>Y</italic> corresponds to a sample, with columns in <italic>X</italic> representing features and columns in <italic>Y</italic> representing binary labels indicating the presence or absence of specific labels. These matrices serve as inputs for the inner optimization of DHBCPSO-MSR to construct similarity graphs and optimize FS.</p>
<p><bold>Step 2: Constructing the sample similarity graph.</bold> To preserve the local geometry of the data, we first construct the sample similarity graph. The similarity between samples is measured using a Gaussian heat kernel. The sample similarity matrix <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:mi>S</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is computed as shown in <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref>.
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left left" rowspacing=".2em" columnspacing="1em" displaystyle="false"><mml:mtr><mml:mtd><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mfrac><mml:mrow><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:mtext>if&#xA0;</mml:mtext></mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x2130;</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mtext>&#xA0;or&#xA0;</mml:mtext></mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x2130;</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:mtext>otherwise</mml:mtext></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:math></disp-formula>where <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represents the similarity between samples <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:msub><mml:mi>x</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula> denotes the squared Euclidean distance between the <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mi>i</mml:mi></mml:math></inline-formula>-th and <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mi>j</mml:mi></mml:math></inline-formula>-th samples, and <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula> is the width parameter of the Gaussian kernel, controlling the decay of similarity. The set <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:msub><mml:mrow><mml:mi>&#x2130;</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represents the <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:mi>k</mml:mi></mml:math></inline-formula>-nearest neighbors of sample <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula>, indicating which samples are considered similar.</p>
<p>Next, we construct the Laplacian matrix <italic>L</italic> &#x003D; <italic>D</italic> &#x2212; <italic>S</italic>, where <italic>D</italic> is the degree matrix, and the diagonal entries of <italic>D</italic> represent the degree of each sample (the number of connections to other samples).</p>
<p>The sample manifold regularization term <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:msub><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mrow><mml:mtext>manifold</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> is then computed as shown in <xref ref-type="disp-formula" rid="eqn-2">Eq. (2)</xref>.
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:msub><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mtext>manifold</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mtext>Tr</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>F</mml:mi><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:mi>F</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>q</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is the embedding matrix that maps the samples <italic>X</italic> to the low-dimensional label space, <italic>L</italic> is the Laplacian matrix derived from the sample similarity graph, and <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:mtext>Tr</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is the trace operator, which sums the diagonal elements of the matrix.</p>
<p><bold>Step 3: Constructing the label similarity graph.</bold> To capture the global dependencies between labels, we construct the label similarity graph. The label similarity matrix <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:msub><mml:mi>S</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>q</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is computed using the Gaussian kernel as shown in <xref ref-type="disp-formula" rid="eqn-3">Eq. (3)</xref>.
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:msub><mml:mi>S</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>=</mml:mo><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mfrac><mml:mrow><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:msub><mml:mi>S</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> is the similarity between labels <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:msub><mml:mi>y</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula> is the squared Euclidean distance between the <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mi>i</mml:mi></mml:math></inline-formula>-th and <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mi>j</mml:mi></mml:math></inline-formula>-th labels, and <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula> is the width parameter of the Gaussian kernel, controlling the decay of similarity.</p>
<p>We then construct the Laplacian matrix <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:msub><mml:mi>L</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:math></inline-formula>, where <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:msub><mml:mi>D</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:math></inline-formula> is the degree matrix of the label similarity graph.</p>
<p>The label manifold regularization term <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:msub><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mrow><mml:mtext>label</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> is calculated as shown in <xref ref-type="disp-formula" rid="eqn-4">Eq. (4)</xref>.
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:msub><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mtext>label</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mtext>Tr</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>L</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:msup><mml:mi>F</mml:mi><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:msub><mml:mi>L</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:math></inline-formula> is the Laplacian matrix derived from the label similarity graph, <italic>F</italic> is the embedding matrix that represents the label space, and <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:mtext>Tr</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is the trace operator.</p>
<p><bold>Step 4: Initialization of feature selection weight matrix <italic>W</italic>. </bold>The FS weight matrix <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:mi>W</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>q</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is used to map features to the label space, aligning features and labels for optimal learning. We initialize <italic>W</italic> using random initialization. Specifically, each element <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is drawn from a uniform distribution <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:mrow><mml:mi>&#x1D4B0;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03F5;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03F5;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, where <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:mi>&#x03F5;</mml:mi></mml:math></inline-formula> is a small constant (e.g., <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:mi>&#x03F5;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.01</mml:mn></mml:math></inline-formula>) to ensure that the initial values of the weights are small enough to avoid instability during optimization as shown in <xref ref-type="disp-formula" rid="eqn-5">Eq. (5)</xref>.
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:mi>&#x1D4B0;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03F5;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03F5;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>The <italic>W</italic> is then regularized using the <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:msub><mml:mi>&#x2113;</mml:mi><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>-norm to enforce sparsity as shown in <xref ref-type="disp-formula" rid="eqn-6">Eq. (6)</xref>.<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:msub><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mtext>sparse</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mi>W</mml:mi><mml:msub><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>d</mml:mi></mml:munderover><mml:msqrt><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>q</mml:mi></mml:munderover><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup></mml:msqrt><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the element in the <inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mi>i</mml:mi></mml:math></inline-formula>-th row and <inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:mi>j</mml:mi></mml:math></inline-formula>-th column of the matrix <italic>W</italic>, representing the weight of the <inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:mi>i</mml:mi></mml:math></inline-formula>-th feature for the <inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:mi>j</mml:mi></mml:math></inline-formula>-th label, and <inline-formula id="ieqn-50"><mml:math id="mml-ieqn-50"><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mi>W</mml:mi><mml:msub><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is the <inline-formula id="ieqn-51"><mml:math id="mml-ieqn-51"><mml:msub><mml:mi>&#x2113;</mml:mi><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>-norm regularization, which promotes row sparsity in <italic>W</italic>, forcing some features to have zero weights and thus performing FS.</p>
<p><bold>Step 5: Combined objective function.</bold> The overall optimization problem combines the loss, manifold regularization, label correlation regularization, and sparsity regularization terms. The combined objective function is shown in <xref ref-type="disp-formula" rid="eqn-7">Eq. (7)</xref>.<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:munder><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>F</mml:mi></mml:mrow></mml:munder><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mi>X</mml:mi><mml:mi>W</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">1</mml:mtext></mml:mrow><mml:mi>n</mml:mi></mml:msub><mml:msup><mml:mi>b</mml:mi><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mo>&#x2212;</mml:mo><mml:mi>F</mml:mi><mml:msubsup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mi>F</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mtext>Tr</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>F</mml:mi><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mtext>Tr</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>L</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:msup><mml:mi>F</mml:mi><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mi>W</mml:mi><mml:msub><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-52"><mml:math id="mml-ieqn-52"><mml:mi>X</mml:mi><mml:mi>W</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">1</mml:mtext></mml:mrow><mml:mi>n</mml:mi></mml:msub><mml:msup><mml:mi>b</mml:mi><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup></mml:math></inline-formula> is the predicted matrix, representing the feature projection into the label space, <italic>F</italic> is the embedding matrix representing the label space representation of the samples, <inline-formula id="ieqn-53"><mml:math id="mml-ieqn-53"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">1</mml:mtext></mml:mrow><mml:mi>n</mml:mi></mml:msub></mml:math></inline-formula> is a vector of ones of length <inline-formula id="ieqn-54"><mml:math id="mml-ieqn-54"><mml:mi>n</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-55"><mml:math id="mml-ieqn-55"><mml:mi>b</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mi>q</mml:mi></mml:msup></mml:math></inline-formula> is the bias vector, and <inline-formula id="ieqn-56"><mml:math id="mml-ieqn-56"><mml:mi>&#x03B1;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03BB;</mml:mi></mml:math></inline-formula> are regularization parameters controlling the importance of manifold regularization, label correlation, and sparsity, respectively.</p>
<p>By minimizing this objective function, the algorithm jointly optimizes the FS weight matrix <italic>W</italic> and the embedding matrix <italic>F</italic>, selecting the most relevant features while preserving both sample and label dependencies. The optimization of <italic>W</italic> ensures FS, while <italic>F</italic> maps samples to the label space, and their joint optimization effectively captures both feature relevance and label structure.</p>
</sec>
<sec id="s3_1_2">
<label>3.1.2</label>
<title>Outer Optimization Layer of DHBCPSO-MSR</title>
<p>The outer optimization layer of DHBCPSO-MSR employs the HBCPSO algorithm to further optimize the FS weight matrix <italic>W</italic> obtained from the inner optimization layer. Based on the theory of heterosis, the HBCPSO algorithm adopts a three-line (maintainer, sterile, and restorer) co-evolution mechanism from HBO. Initially, the population is ranked based on fitness (calculated by <xref ref-type="disp-formula" rid="eqn-7">Eq. (7)</xref>) and divided into three subgroups: the maintainer, sterile, and restorer lines. A PSO algorithm is then applied within each subgroup for global updates, including both velocity and position adjustments. After a fixed number of iterations, the best solutions from each subgroup exchange information periodically, guiding further optimization within the subgroups. Once the outer optimization layer updates the feature weight matrix <inline-formula id="ieqn-57"><mml:math id="mml-ieqn-57"><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mtext>updated</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>, the result is fed back into the inner optimization layer as part of the objective function to assess convergence.</p>
</sec>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Federated Multi-Label Feature Selection via DHBCPSO-MSR (Fed-MFSDHBCPSO)</title>
<p>Centralized MFS methods face significant challenges in distributed environments: centralized approaches risk privacy breaches, while federated methods rely on FL&#x2019;s inherent privacy guarantees, which are often insufficient. Moreover, limited data sharing hinders the capture of inter-feature correlations, causing convergence to the local optima and limiting the global performance of FS. To address these issues, we propose Fed-MFSDHBCPSO&#x2014;a FL framework that integrates DHBCPSO-MSR, DP, and SMPC, enhanced with manifold regularization and sparsity constraints (see <xref ref-type="fig" rid="fig-2">Fig. 2</xref>).</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>The architecture of the proposed Fed-MFSDHBCPSO: (<bold>a</bold>) Encrypt the optimal feature subset using DP. (<bold>b</bold>) Send the encrypted indices and weights to the server. (<bold>c</bold>) The server applies SMPC to securely aggregate and share encrypted results until global convergence, generating the global optimal feature weights and indices. (<bold>d</bold>) Distribute the aggregated weights back to clients</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-2.tif"/>
</fig>
<p><italic>1) Client-side MFS process</italic></p>
<p>Each client performs local FS using the following the following steps:</p>
<p><bold>Step 1: Local data preprocessing.</bold> Clients load local datasets, initialize parameters, and perform data cleaning and normalization to ensure high-quality inputs.</p>
<p><bold>Step 2: Local FS via DHBCPSO-MSR.</bold> Each client executes DHBCPSO-MSR, integrating an inner modeling layer with manifold construction and sparsity constraints, alongside an outer optimization layer driven by HBCPSO. This dual-layer algorithm ensures optimal feature subset selection that aligns with the client&#x2019;s local data distribution.</p>
<p><bold>Step 3: Privacy-preserving transmission.</bold> To ensure privacy, the selected feature weights and their corresponding indices are perturbed using a differential privacy (DP) mechanism before being transmitted to the central server, thereby achieving <inline-formula id="ieqn-58"><mml:math id="mml-ieqn-58"><mml:mi>&#x03B5;</mml:mi></mml:math></inline-formula>-DP as defined in <xref ref-type="disp-formula" rid="eqn-8">Eq. (8)</xref>. Specifically, Gaussian noise is added to each feature weight to obscure the influence of individual data samples and defend against potential inference attacks. The perturbation process is given by:<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:msub><mml:mrow><mml:mover><mml:mi>W</mml:mi><mml:mo>&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mrow><mml:mi>&#x1D4A9;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-59"><mml:math id="mml-ieqn-59"><mml:msub><mml:mrow><mml:mover><mml:mi>W</mml:mi><mml:mo>&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> denotes the perturbed feature weight matrix for the <inline-formula id="ieqn-60"><mml:math id="mml-ieqn-60"><mml:mi>i</mml:mi></mml:math></inline-formula>-th feature, and <inline-formula id="ieqn-61"><mml:math id="mml-ieqn-61"><mml:mrow><mml:mi>&#x1D4A9;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represents Gaussian noise with zero mean and variance <inline-formula id="ieqn-62"><mml:math id="mml-ieqn-62"><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula>. The noise scale <inline-formula id="ieqn-63"><mml:math id="mml-ieqn-63"><mml:mi>&#x03C3;</mml:mi></mml:math></inline-formula> is calibrated based on the desired privacy level using the standard <inline-formula id="ieqn-64"><mml:math id="mml-ieqn-64"><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x03B5;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03B4;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>-DP framework. In this paper, we follow common practices in the literature and set the privacy budget to <inline-formula id="ieqn-65"><mml:math id="mml-ieqn-65"><mml:mi>&#x03B5;</mml:mi><mml:mo>=</mml:mo><mml:mn>1.0</mml:mn></mml:math></inline-formula>, which achieves a good balance between privacy preservation and model utility [<xref ref-type="bibr" rid="ref-21">21</xref>]. Smaller values of <inline-formula id="ieqn-66"><mml:math id="mml-ieqn-66"><mml:mi>&#x03B5;</mml:mi></mml:math></inline-formula> (e.g., &#x003C;1) provide stronger privacy but may degrade performance, whereas larger values (e.g., &#x003E;5) weaken privacy guarantees. By setting <inline-formula id="ieqn-67"><mml:math id="mml-ieqn-67"><mml:mi>&#x03B5;</mml:mi><mml:mo>=</mml:mo><mml:mn>1.0</mml:mn></mml:math></inline-formula>, we align with the commonly accepted range <inline-formula id="ieqn-68"><mml:math id="mml-ieqn-68"><mml:mo stretchy="false">[</mml:mo><mml:mn>0.5</mml:mn><mml:mo>,</mml:mo><mml:mn>5</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> in FL scenarios, ensuring both practical utility and privacy protection.</p>
<p><bold>Step 4: Local multi-label classification.</bold> Using the refined features, each client applies the multi-label k-nearest neighbors (ML-KNN) [<xref ref-type="bibr" rid="ref-19">19</xref>] for local multi-label classification. This decentralized approach minimizes communication overhead while ensuring data privacy.</p>
<p><italic>2) Server-side dynamic collaborative optimization</italic></p>
<p>The server implements a dynamic collaborative optimization strategy comprising the following steps:</p>
<p><bold>Step 1: Key management.</bold> A public key infrastructure (PKI) is employed to securely distribute encryption keys, enabling end-to-end secure communication across the federated network.</p>
<p><bold>Step 2: Feature aggregation and distribution.</bold></p>
<p>(i) <bold>Secure aggregation via SMPC:</bold> Encrypted feature weights from clients are aggregated using SMPC, which ensures data privacy by allowing the server to compute aggregated feature weights without decryption. In this study, the SMPC protocol is simulated through a lightweight, custom implementation on the MATLAB platform. While standard secure computation frameworks such as CrypTen or PySyft are not utilized, our implementation adheres to core SMPC principles, enabling secure aggregation without decryption. This ensures that the server remains unable to access any client&#x2019;s plaintext feature weights or index information during the aggregation process. The prototype is designed to evaluate the feasibility and computational efficiency of secure aggregation in resource-constrained edge computing environments.</p>
<p>(ii) <bold>Global feature weight computation:</bold> The global feature weight vector <inline-formula id="ieqn-69"><mml:math id="mml-ieqn-69"><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mtext>global</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> is computed as in <xref ref-type="disp-formula" rid="eqn-9">Eq. (9)</xref>.<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mrow><mml:mtext>global</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:msup><mml:mi>W</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>c</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-70"><mml:math id="mml-ieqn-70"><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mtext>global</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> represents the aggregated feature importance vector across all clients, and <inline-formula id="ieqn-71"><mml:math id="mml-ieqn-71"><mml:msup><mml:mi>W</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>c</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula> is the sparse, encrypted vector from each client. The weighting factor <inline-formula id="ieqn-72"><mml:math id="mml-ieqn-72"><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula> is calculated based on the dataset size: <inline-formula id="ieqn-73"><mml:math id="mml-ieqn-73"><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:math></inline-formula>.</p>
<p><bold>Step 3: Global iterative optimization.</bold> The server iteratively aggregates encrypted feature weights using SMPC to update <inline-formula id="ieqn-74"><mml:math id="mml-ieqn-74"><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mtext>global</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>, ensuring data confidentiality without accessing plaintext data, and sends the updated model back to clients for further iterations.</p>
<p>In this framework, clients apply DP to perturb their feature weights for privacy, while the server aggregates encrypted feature weights securely using SMPC. This process ensures that no sensitive data is exposed during aggregation.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Further Analysis</title>
<p>To thoroughly evaluate the practical applicability of the proposed method, we analyze the privacy preservation capability and computational complexity of Fed-MFSDHBCPSO.</p>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>Privacy Preservation Capability</title>
<p>As illustrated in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>, the server communicates exclusively with individual clients, preventing any direct client-to-client interaction. Throughout this process, feature weight results are exchanged while maintaining the confidentiality of sample data. Following the design of FL frameworks like FATE [<xref ref-type="bibr" rid="ref-34">34</xref>], communication is strictly confined to model parameters, ensuring that clients cannot access each other&#x2019;s data distributions and thereby strengthening privacy protection. When feature names and their combinations are non-sensitive&#x2014;for example, hospitals sharing disease-related attributes without disclosing individual patient information&#x2014;Fed-MFSDHBCPSO effectively safeguards privacy. Clients may employ DP to secure data transmission, and the server utilizes SMPC to aggregate encrypted feature weights without decrypting them. Detailed cryptographic implementations are beyond the scope of this paper and will not be discussed further.</p>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>Communication Analysis</title>
<p>In the FL framework, Fed-MFSDHBCPSO ensures data privacy by restricting each communication round to essential interactions between the server and clients. Each round consists of three steps:</p>
<p>Step 1: Each client uploads the encrypted feature indices along with the weights of its locally selected optimal subset, avoiding the transmission of raw data or model parameters to ensure privacy and reduce communication overhead;</p>
<p>Step 2: The server aggregates the weights received from all clients and broadcasts the aggregated result to them;</p>
<p>Step 3: Clients perform local optimization based on the aggregated information and upload the encrypted updates back to the server.</p>
<p>Only <italic>D</italic>-dimensional indices and a small number of weight values are transmitted per round, resulting in low communication overhead relative to local computation. Moreover, since Fed-MFSDHBCPSO performs feature subset selection and compression locally at each client in advance, only highly condensed information is exchanged globally, further reducing overall communication cost.</p>
</sec>
<sec id="s3_3_3">
<label>3.3.3</label>
<title>Computational Complexity Analysis</title>
<p>The DHBCPSO-MSR algorithm minimizes computational overhead through efficient matrix operations and lightweight data structures, ensuring robust performance even on resource-constrained devices. The HBCPSO component further improves efficiency by leveraging parallel computation and dynamic task allocation tailored to device hardware. We utilize MATLAB&#x2019;s <monospace>parfor</monospace> and <monospace>parpool</monospace> functions to parallelize particle evaluations, significantly accelerating the optimization process. Within the FL framework, all computations are performed locally on clients, reducing both data transmission and central processing overhead. To safeguard privacy, only feature weights are encrypted, further minimizing communication costs.</p>
<p>In Fed-MFSDHBCPSO, the total time complexity comprises both client-side and server-side components. On the client side, the inner modeling layer has a complexity of <inline-formula id="ieqn-75"><mml:math id="mml-ieqn-75"><mml:mi>O</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, where <inline-formula id="ieqn-76"><mml:math id="mml-ieqn-76"><mml:mi>n</mml:mi></mml:math></inline-formula> is the number of samples. The outer HBCPSO optimization contributes <inline-formula id="ieqn-77"><mml:math id="mml-ieqn-77"><mml:mi>O</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>T</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, with <italic>N</italic> particles, <italic>T</italic> iterations, and <inline-formula id="ieqn-78"><mml:math id="mml-ieqn-78"><mml:mi>d</mml:mi></mml:math></inline-formula> features. Differential privacy perturbation introduces an additional <inline-formula id="ieqn-79"><mml:math id="mml-ieqn-79"><mml:mi>O</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> cost per client. On the server side, secure aggregation and global weight computation using SMPC require <inline-formula id="ieqn-80"><mml:math id="mml-ieqn-80"><mml:mi>O</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, while sorting incurs <inline-formula id="ieqn-81"><mml:math id="mml-ieqn-81"><mml:mi>O</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>d</mml:mi><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, where <italic>K</italic> is the number of clients. The overall complexity is:<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mi>O</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>O</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>T</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>O</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>d</mml:mi><mml:mo>+</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>d</mml:mi><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>demonstrating good scalability with respect to the number of clients and features. Although the client-side <inline-formula id="ieqn-82"><mml:math id="mml-ieqn-82"><mml:mi>O</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> complexity may raise concerns for large-scale datasets, it is mitigated through batch processing and parallelization. Since this step is executed locally and only once per round, its overall impact remains limited.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experiments</title>
<p>In this section, we describe the experimental setup, compare the proposed Fed-MFSDHBCPSO with both centralized and federated approaches, and conduct a statistical significance analysis.</p>
<sec id="s4_1">
<label>4.1</label>
<title>Experimental Settings</title>
<sec id="s4_1_1">
<label>4.1.1</label>
<title>Federated Simulation and Dataset Description</title>
<p>In this study, the FL framework is simulated in a single-machine environment using multiprocessing, where each client independently loads its local data and performs FS. This setup closely mirrors real-world scenarios involving decentralized computation and data isolation.</p>
<p>To evaluate the performance of Fed-MFSDHBCPS, we simulate a FL environment using eight real-world datasets (see <xref ref-type="table" rid="table-1">Table 1</xref>). To emulate non-IID conditions, we adopt a label distribution skew strategy [<xref ref-type="bibr" rid="ref-19">19</xref>], in which samples are grouped by label frequency and allocated to clients through stratified sampling. This approach produces imbalanced and heterogeneous multi-label distributions across clients, effectively capturing the statistical heterogeneity typical of edge devices.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Details of the experimental datasets</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>No.</th>
<th>Dataset</th>
<th>Samples</th>
<th>Features</th>
<th>Labels</th>
<th>Domain</th>
<th>Datatype</th>
</tr>
</thead>
<tbody>
<tr>
<td>1</td>
<td>Flags</td>
<td>194</td>
<td>7</td>
<td>19</td>
<td>Image</td>
<td>Discrete</td>
</tr>
<tr>
<td>2</td>
<td>VirusGo</td>
<td>207</td>
<td>749</td>
<td>6</td>
<td>Biology</td>
<td>Discrete</td>
</tr>
<tr>
<td>3</td>
<td>Emotions</td>
<td>593</td>
<td>72</td>
<td>6</td>
<td>Music</td>
<td>Discrete</td>
</tr>
<tr>
<td>4</td>
<td>Enron</td>
<td>1702</td>
<td>1001</td>
<td>53</td>
<td>Text</td>
<td>Discrete</td>
</tr>
<tr>
<td>5</td>
<td>Image</td>
<td>2000</td>
<td>294</td>
<td>5</td>
<td>Image</td>
<td>Continuous</td>
</tr>
<tr>
<td>6</td>
<td>Scene</td>
<td>2407</td>
<td>294</td>
<td>6</td>
<td>Image</td>
<td>Continuous</td>
</tr>
<tr>
<td>7</td>
<td>Education</td>
<td>5000</td>
<td>550</td>
<td>33</td>
<td>Text</td>
<td>Continuous</td>
</tr>
<tr>
<td>8</td>
<td>Mediamill</td>
<td>43,907</td>
<td>120</td>
<td>101</td>
<td>Video Tagging</td>
<td>Continuous</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_1_2">
<label>4.1.2</label>
<title>Evaluation Metrics</title>
<p>In this paper, we utilize six standard multi-label evaluation metrics. Average Precision (AP) and Macro-F1 (MA) measure overall prediction accuracy (higher values are better), while Coverage (CV), Hamming Loss (HL), and Ranking Loss (RL) evaluate errors and ranking quality (lower values are better). Micro-F1 (MI) assesses global classification performance (higher values are better). AP calculates the average rank of labels for each instance, HL quantifies the discrepancy between the predicted and true label sets, and MA computes the average F1 score based on true positives, false positives, and false negatives for each instance. Further details on the definitions and computation methods are provided in [<xref ref-type="bibr" rid="ref-29">29</xref>].</p>
</sec>
<sec id="s4_1_3">
<label>4.1.3</label>
<title>Comparison Algorithms and Parameter Settings</title>
<p>We compare the proposed method with various centralized and federated MFS approaches. Recent federated MFS methods include FMLFS [<xref ref-type="bibr" rid="ref-18">18</xref>], Fuzzy FMFS [<xref ref-type="bibr" rid="ref-19">19</xref>], and FedCMFS [<xref ref-type="bibr" rid="ref-20">20</xref>], while centralized methods comprise deep learning-based MFS approaches such as LRDG [<xref ref-type="bibr" rid="ref-30">30</xref>] (referred to as Fed-LRDG in the federated setting), multi-objective optimization algorithms like MOEA/D [<xref ref-type="bibr" rid="ref-35">35</xref>] and NSGA-III [<xref ref-type="bibr" rid="ref-36">36</xref>] (referred to as Fed-MOEA/D and Fed-NSGA-III in the federated setting), as well as GLFS [<xref ref-type="bibr" rid="ref-37">37</xref>] and PDMFS [<xref ref-type="bibr" rid="ref-38">38</xref>]. Under federated learning settings, centralized methods are independently applied to each client, and results are averaged across clients, highlighting the limitations of centralized MFS and underscoring the advantages of federated methods in terms of privacy and collaboration. Parameter settings for these methods follow those reported in the original literature.</p>
<p>For Fed-MFSDHBCPSO, we adopt the parameter settings from [<xref ref-type="bibr" rid="ref-13">13</xref>]: population size &#x003D; 30,100 iterations, and 30 runs. In the outer HBCPSO layer, the inertia weight decreases linearly from 0.9 to 0.2, with cognitive and social factors both set to 2.0, and maximum velocity set to 1.0. For the inner-layer optimization in <xref ref-type="disp-formula" rid="eqn-7">Eq. (7)</xref>, the parameters are <inline-formula id="ieqn-83"><mml:math id="mml-ieqn-83"><mml:mi>&#x03B1;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn></mml:math></inline-formula>, <inline-formula id="ieqn-84"><mml:math id="mml-ieqn-84"><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn></mml:math></inline-formula>, and <inline-formula id="ieqn-85"><mml:math id="mml-ieqn-85"><mml:mi>&#x03B3;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.1</mml:mn></mml:math></inline-formula>, as detailed in <xref ref-type="sec" rid="s4_2">Section 4.2</xref>.</p>
</sec>
<sec id="s4_1_4">
<label>4.1.4</label>
<title>Experimental Environment</title>
<p>All experiments are conducted on a desktop computer running the Windows 10 operating system, equipped with a 13th Gen Intel(R) Core(TM) i7-13700KF processor @ 3.40GHz and 32 GB of memory. The software environment used for the experiments is MATLAB 2022.</p>
</sec>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Parameter Sensitivity Analysis</title>
<p>We conducte experiments with five clients and three datasets of varying sizes (Flags, Education, and Mediamill), selecting the final number of features according to the method in [<xref ref-type="bibr" rid="ref-39">39</xref>]. A sensitivity analysis of the hyperparameters <bold><inline-formula id="ieqn-86"><mml:math id="mml-ieqn-86"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula></bold>, <bold><inline-formula id="ieqn-87"><mml:math id="mml-ieqn-87"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula></bold>, and <bold><inline-formula id="ieqn-88"><mml:math id="mml-ieqn-88"><mml:mi>&#x03BB;</mml:mi></mml:math></inline-formula></bold> is performed to evaluate their impact on the method&#x2019;s performance. Different parameter settings are tested by varying one hyperparameter at a time, and the model&#x2019;s performance is assessed using AP and HL. The results, shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, demonstrate that regularization weights significantly influence feature sparsity, label structure preservation, and model convergence. The optimal hyperparameter combination of [0.5, 0.5, 0.1] is identified, providing a strong foundation for further optimization and application.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Parameter sensitivity analysis of the proposed method: (<bold>a</bold>) The impact of different <bold><inline-formula id="ieqn-89"><mml:math id="mml-ieqn-89"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula></bold> values on average precision and hamming loss; (<bold>b</bold>) The impact of different <bold><inline-formula id="ieqn-90"><mml:math id="mml-ieqn-90"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula></bold> values on average precision and hamming loss; (<bold>c</bold>) The impact of different <bold><inline-formula id="ieqn-91"><mml:math id="mml-ieqn-91"><mml:mi>&#x03BB;</mml:mi></mml:math></inline-formula></bold> values on average precision and hamming loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-3.tif"/>
</fig>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Comparative Analysis of Fed-MFSDHBCPSO with Centralized and Federated Approaches</title>
<p>To validate the effectiveness of the proposed method, <xref ref-type="fig" rid="fig-4">Figs. 4</xref>&#x2013;<xref ref-type="fig" rid="fig-11">11</xref> compare it against the initial baseline, five recent centralized methods, and three federated MFS methods across eight datasets. The experiments are conducted with five clients, assessing performance across varying feature subsets. The results show that Fed-MFSDHBCPSO outperforms other methods on most datasets with only 10 features, including Mediamill, Image, Scene, Flag, and Education. On the Enron and Emotions datasets, Fed-MFSDHBCPSO, using 80 features, achieves comparable or superior Macro-F1, Micro-F1, and Hamming loss scores, while demonstrating more consistent performance across other metrics.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Performance comparison of different methods on the Flags dataset with varying numbers of features based on six metrics: (<bold>a</bold>) Average precision; (<bold>b</bold>) Coverage; (<bold>c</bold>) Macro-F1; (<bold>d</bold>) Micro-F1; (<bold>e</bold>) Hamming loss; (<bold>f</bold>) Ranking loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-4.tif"/>
</fig><fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Performance comparison of different methods on the VirusGo dataset with varying numbers of features based on six metrics: (<bold>a</bold>) Average precision; (<bold>b</bold>) Coverage; (<bold>c</bold>) Macro-F1; (<bold>d</bold>) Micro-F1; (<bold>e</bold>) Hamming loss; (<bold>f</bold>) Ranking loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-5.tif"/>
</fig><fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Performance comparison of different methods on the Emotions dataset with varying numbers of features based on six metrics: (<bold>a</bold>) Average precision; (<bold>b</bold>) Coverage; (<bold>c</bold>) Macro-F1; (<bold>d</bold>) Micro-F1; (<bold>e</bold>) Hamming loss; (<bold>f</bold>) Ranking loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-6.tif"/>
</fig><fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Performance comparison of different methods on the Enron dataset with varying numbers of features based on six metrics: (<bold>a</bold>) Average precision; (<bold>b</bold>) Coverage; (<bold>c</bold>) Macro-F1; (<bold>d</bold>) Micro-F1; (<bold>e</bold>) Hamming loss; (<bold>f</bold>) Ranking loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-7.tif"/>
</fig><fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Performance comparison of different methods on the Image dataset with varying numbers of features based on six metrics: (<bold>a</bold>) Average precision; (<bold>b</bold>) Coverage; (<bold>c</bold>) Macro-F1; (<bold>d</bold>) Micro-F1; (<bold>e</bold>) Hamming loss; (<bold>f</bold>) Ranking loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-8.tif"/>
</fig><fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>Performance comparison of different methods on the Scene dataset with varying numbers of features based on six metrics: (<bold>a</bold>) Average precision; (<bold>b</bold>) Coverage; (<bold>c</bold>) Macro-F1; (<bold>d</bold>) Micro-F1; (<bold>e</bold>) Hamming loss; (<bold>f</bold>) Ranking loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-9.tif"/>
</fig><fig id="fig-10">
<label>Figure 10</label>
<caption>
<title>Performance comparison of different methods on the Education dataset with varying numbers of features based on six metrics: (<bold>a</bold>) Average precision; (<bold>b</bold>) Coverage; (<bold>c</bold>) Macro-F1; (<bold>d</bold>) Micro-F1; (<bold>e</bold>) Hamming loss; (<bold>f</bold>) Ranking loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-10.tif"/>
</fig><fig id="fig-11">
<label>Figure 11</label>
<caption>
<title>Performance comparison of different methods on the Mediamill dataset with varying numbers of features based on six metrics: (<bold>a</bold>) Average precision; (<bold>b</bold>) Coverage; (<bold>c</bold>) Macro-F1; (<bold>d</bold>) Micro-F1; (<bold>e</bold>) Hamming loss; (<bold>f</bold>) Ranking loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-11.tif"/>
</fig>
<p>These results indicate that the proposed method has minimal impact on communication overhead and model performance, despite the size of the feature set transmitted between clients and the edge server. It simultaneously enhances the effectiveness of the model, achieving an optimal balance between communication efficiency and predictive accuracy.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Comparison of Running Time for Different Methods</title>
<p>To evaluate the efficiency of the proposed method, we conducted a single federated iteration with five clients and compared the runtime of Fed-MFSDHBCPSO against several baseline methods, as presented in <xref ref-type="table" rid="table-2">Table 2</xref>, where the bold values represent the best results. The results demonstrate that Fed-MFSDHBCPSO consistently achieves the shortest runtimes, particularly on large-scale datasets such as Mediamill, Enron, and Education. This superior efficiency stems from its dual-layer optimization, manifold regularization, sparsity constraints, and FL framework. In contrast, the Org method, which lacks FS, leads to significantly longer runtimes. Furthermore, methods such as Fed-LRDG, Fed-MOEA/D, Fed-NSGA-III, and PDMFS exhibit higher computational overhead on large datasets. Overall, Fed-MFSDHBCPSO delivers the best runtime efficiency among all evaluated algorithms.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Comparison of running time for different methods</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Dataset</th>
<th>Proposed</th>
<th>Org</th>
<th>FedCMFS</th>
<th>FMLFS</th>
<th align="center">Fuzzy<break/>FMFS</th>
<th align="center">Fed-<break/>LRDG</th>
<th align="center">Fed-<break/>MOEA/D</th>
<th align="center">Fed-<break/>NSGA-<break/>III</th>
<th>GLFS</th>
<th>PDMFS</th>
</tr>
</thead>
<tbody>
<tr>
<td>Flags</td>
<td><bold>0.4918</bold></td>
<td>0.86</td>
<td>0.75</td>
<td>0.65</td>
<td>0.78</td>
<td>0.85</td>
<td>0.91</td>
<td>0.80</td>
<td>0.92</td>
<td>1.05</td>
</tr>
<tr>
<td>VirusGo</td>
<td><bold>3.3264</bold></td>
<td>5.90</td>
<td>4.45</td>
<td>4.20</td>
<td>4.65</td>
<td>5.00</td>
<td>5.15</td>
<td>4.60</td>
<td>5.30</td>
<td>5.60</td>
</tr>
<tr>
<td>Emotions</td>
<td><bold>0.5958</bold></td>
<td>0.965</td>
<td>0.85</td>
<td>0.75</td>
<td>0.88</td>
<td>0.95</td>
<td>1.00</td>
<td>0.90</td>
<td>1.75</td>
<td>1.15</td>
</tr>
<tr>
<td>Enron</td>
<td><bold>12.34</bold></td>
<td>15.00</td>
<td>15.80</td>
<td>14.00</td>
<td>16.20</td>
<td>17.50</td>
<td>18.00</td>
<td>17.00</td>
<td>20.50</td>
<td>19.20</td>
</tr>
<tr>
<td>Image</td>
<td><bold>1.5487</bold></td>
<td>2.10</td>
<td>2.10</td>
<td>1.85</td>
<td>2.20</td>
<td>2.50</td>
<td>2.60</td>
<td>2.30</td>
<td>3.20</td>
<td>2.90</td>
</tr>
<tr>
<td>Scene</td>
<td><bold>1.7899</bold></td>
<td>2.80</td>
<td>2.30</td>
<td>2.00</td>
<td>2.40</td>
<td>2.60</td>
<td>2.80</td>
<td>2.50</td>
<td>3.10</td>
<td>3.10</td>
</tr>
<tr>
<td>Education</td>
<td><bold>16.008</bold></td>
<td>23.50</td>
<td>19.80</td>
<td>18.00</td>
<td>20.10</td>
<td>21.50</td>
<td>22.00</td>
<td>21.00</td>
<td>22.50</td>
<td>23.20</td>
</tr>
<tr>
<td>Mediamill</td>
<td><bold>100.57</bold></td>
<td>152.10</td>
<td>132.10</td>
<td>126.85</td>
<td>133.30</td>
<td>140.50</td>
<td>143.60</td>
<td>148.40</td>
<td>138.70</td>
<td>153.90</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Ablation Study</title>
<p>To evaluate the contribution of each component, we conducted ablation experiments on five clients, determining the final number of selected features using the method in [<xref ref-type="bibr" rid="ref-39">39</xref>]. Four model variants were constructed by individually removing PSO (Proposed/PSO), HBO (Proposed/HBO), manifold regularization (i.e., removing Laplacian regularization, called Proposed/MR), and the sparsity constraint (Proposed/SC). In addition, we removed the inner optimization to create Fed-HBCPSO and compared it with Fed-NSGA-III and Fed-MOEA/D, using AP as the primary metric. 
As shown in <xref ref-type="table" rid="table-3">Table 3</xref>, where the bold values indicate the best results. Excluding PSO or HBO led to significant performance degradation, while removing manifold regularization or the sparsity constraint also reduced accuracy and stability. Moreover, Fed-HBCPSO consistently outperformed Fed-NSGA-III and Fed-MOEA/D, benefiting from its three-population co-evolutionary mechanism, where sub-populations of varying quality periodically exchange best-solution information. This design effectively balances exploration and exploitation, avoids the local optima, and accelerates convergence. In contrast, the single-population strategies of NSGA-III and MOEA/D in federated settings struggle to maintain both diversity and precision, resulting in inferior performance. These findings demonstrate that all components are complementary and indispensable to the overall effectiveness of the proposed model.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Comparison of the proposed method with five ablated variants and two baseline methods</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Dataset</th>
<th>Proposed</th>
<th align="center">Proposed/<break/>PSO</th>
<th align="center">Proposed/<break/>HBO</th>
<th align="center">Proposed/<break/>MR</th>
<th align="center">Proposed/<break/>SC</th>
<th align="center">Fed-<break/>HBCPSO</th>
<th align="center">Fed-<break/>NSGA-III</th>
<th align="center">Fed-<break/>MOEA/D</th>
</tr>
</thead>
<tbody>
<tr>
<td>Flags</td>
<td><bold>77.83%</bold></td>
<td>70.02%</td>
<td>69%</td>
<td>68.88%</td>
<td>68.76%</td>
<td>73.81%</td>
<td>71.26%</td>
<td>71.03%</td>
</tr>
<tr>
<td>VirusGo</td>
<td><bold>78.84%</bold></td>
<td>71.12%</td>
<td>70.01%</td>
<td>69.89%</td>
<td>69.57%</td>
<td>74.46%</td>
<td>72.17%</td>
<td>71.23%</td>
</tr>
<tr>
<td>Emotions</td>
<td><bold>72.03%</bold></td>
<td>61.02%</td>
<td>62.12%</td>
<td>63.18%</td>
<td>63.32%</td>
<td>65.56%</td>
<td>63.39%</td>
<td>63.75%</td>
</tr>
<tr>
<td>Enron</td>
<td><bold>62.13%</bold></td>
<td>52.56%</td>
<td>53.71%</td>
<td>56.13%</td>
<td>57.00%</td>
<td>59.35%</td>
<td>57.61%</td>
<td>56.89%</td>
</tr>
<tr>
<td>Image</td>
<td><bold>79.43%</bold></td>
<td>72.13%</td>
<td>73.12%</td>
<td>75.23%</td>
<td>76.12%</td>
<td>76.85%</td>
<td>74.32%</td>
<td>75.25%</td>
</tr>
<tr>
<td>Scene</td>
<td><bold>83.12%</bold></td>
<td>73.21%</td>
<td>72.23%</td>
<td>77.03%</td>
<td>76.56%</td>
<td>79.58%</td>
<td>79.25%</td>
<td>77.42%</td>
</tr>
<tr>
<td>Education</td>
<td><bold>61.01%</bold></td>
<td>55.06%</td>
<td>54.01%</td>
<td>57.87%</td>
<td>56.23%</td>
<td>57.27%</td>
<td>56.38%</td>
<td>56.09%</td>
</tr>
<tr>
<td>Mediamill</td>
<td><bold>72.94%</bold></td>
<td>65.04%</td>
<td>66.32%</td>
<td>67.35%</td>
<td>67.54%</td>
<td>68.96%</td>
<td>66.64%</td>
<td>65.61%</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_6">
<label>4.6</label>
<title>Statistical Analysis of Significance</title>
<p>To assess the performance differences between Fed-MFSDHBCPSO and other comparison methods, we apply the Friedman test followed by the Nemenyi post-hoc test. As shown in <xref ref-type="table" rid="table-4">Table 4</xref>, the Friedman statistics (<inline-formula id="ieqn-92"><mml:math id="mml-ieqn-92"><mml:msub><mml:mi>F</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula>) for all evaluation metrics exceed the critical value of 2.0320 at the 0.05 significance level, and the corresponding <inline-formula id="ieqn-93"><mml:math id="mml-ieqn-93"><mml:mi>p</mml:mi></mml:math></inline-formula>-values are all below 0.05. These results strongly reject the null hypothesis of no significant performance difference, indicating that Fed-MFSDHBCPSO demonstrates statistically significant improvements compared to the other methods.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Friedman test <inline-formula id="ieqn-94"><mml:math id="mml-ieqn-94"><mml:msub><mml:mi>F</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula> for <inline-formula id="ieqn-95"><mml:math id="mml-ieqn-95"><mml:mi>k</mml:mi></mml:math></inline-formula> &#x003D; 9, <italic>N</italic> &#x003D; 8, critical value, and <inline-formula id="ieqn-96"><mml:math id="mml-ieqn-96"><mml:mi>p</mml:mi></mml:math></inline-formula>-values at <inline-formula id="ieqn-97"><mml:math id="mml-ieqn-97"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> &#x003D; 0.05</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Metric</th>
<th><inline-formula id="ieqn-98"><mml:math id="mml-ieqn-98"><mml:msub><mml:mi>F</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula></th>
<th>Critical value</th>
<th><inline-formula id="ieqn-99"><mml:math id="mml-ieqn-99"><mml:mi>p</mml:mi></mml:math></inline-formula>-value</th>
</tr>
</thead>
<tbody>
<tr>
<td>Average precision</td>
<td>19.8489</td>
<td rowspan="6">2.0320</td>
<td>0.0109</td>
</tr>
<tr>
<td>Coverage</td>
<td>17.8135</td>

<td>0.0227</td>
</tr>
<tr>
<td>Macro-F1</td>
<td>22.9056</td>

<td>0.0035</td>
</tr>
<tr>
<td>Micro-F1</td>
<td>21.0534</td>

<td>0.0070</td>
</tr>
<tr>
<td>Hamming loss</td>
<td>19.6823</td>

<td>0.0116</td>
</tr>
<tr>
<td>Ranking loss</td>
<td>18.1143</td>

<td>0.0204</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Following this, the Nemenyi post-hoc test is applied to identify which pairs of methods differ significantly. This test compares the mean ranks of the methods, where a lower rank indicates better performance. A significant difference between two methods is confirmed if the gap in their mean ranks exceeds the critical distance (CD), illustrated by the red line in the figure. The CD is computed as shown in <xref ref-type="disp-formula" rid="eqn-11">Eq. (11)</xref>.<disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:mi>C</mml:mi><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x03B1;</mml:mi></mml:msub><mml:msqrt><mml:mfrac><mml:mrow><mml:mi>k</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mn>6</mml:mn><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:msqrt><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-100"><mml:math id="mml-ieqn-100"><mml:mi>k</mml:mi></mml:math></inline-formula> is the number of methods being compared, and <italic>N</italic> is the number of datasets, and <inline-formula id="ieqn-101"><mml:math id="mml-ieqn-101"><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x03B1;</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>3.098</mml:mn></mml:math></inline-formula> corresponds to a significance level of <inline-formula id="ieqn-102"><mml:math id="mml-ieqn-102"><mml:mi>&#x03B1;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn></mml:math></inline-formula>, resulting in <inline-formula id="ieqn-103"><mml:math id="mml-ieqn-103"><mml:mi>C</mml:mi><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mn>3.7943</mml:mn></mml:math></inline-formula>. This threshold is used to determine whether the performance differences between methods are statistically significant.</p>
<p><xref ref-type="fig" rid="fig-12">Fig. 12</xref> presents the results of the Nemenyi test, with the left side highlighting methods with superior performance rankings and the horizontal axis indicating the average ranks of these approaches. The differences in rankings between the proposed Fed-MFSDHBCPSO and FMLFS, PDMFS, and GLFS exceed the CD, demonstrating that Fed-MFSDHBCPSO significantly outperforms these methods. Conversely, the rankings of Fed-CFMS and Fuzzy FMFS lie within the CD threshold, indicating no statistically significant difference from Fed-MFSDHBCPSO. Nevertheless, Fed-MFSDHBCPSO consistently attains the highest rank, while Fed-CFMS and Fuzzy FMFS exhibit comparative.</p>
<fig id="fig-12">
<label>Figure 12</label>
<caption>
<title>Using the Nemenyi test to compare the performance of Fed-MFSDHBCPSO with other comparison methods on the following six metrics: (<bold>a</bold>) Average precision; (<bold>b</bold>) Hamming loss; (<bold>c</bold>) Coverage; (<bold>d</bold>) Macro-F1; (<bold>e</bold>) Micro-F1; (<bold>f</bold>) Ranking loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_68044-fig-12.tif"/>
</fig>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusion</title>
<p>This paper presents Fed-MFSDHBCPSO, a federated multi-label feature selection (MFS) framework based on the DHBCPSO-MSR algorithm. The framework integrates HBCPSO, differential privacy, secure multi-party computation, manifold regularization, and sparsity constraints to address high computational cost, low efficiency, and inadequate privacy protection in high-dimensional multi-label data. Fed-MFSDHBCPSO enables clients to perform local MFS with DHBCPSO-MSR, where the inner layer preserves manifold structure and sparsity via the <inline-formula id="ieqn-104"><mml:math id="mml-ieqn-104"><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>-norm, while the outer optimization layer leverages HBCPSO, combining HBO&#x2019;s three-population co-evolutionary strategy with PSO&#x2019;s global search for efficient optimization. DP and SMPC further secure the aggregation of encrypted feature subsets on the server side. Experiments on multiple real-world multi-label datasets show that Fed-MFSDHBCPSO consistently outperforms leading centralized and federated methods across metrics such as AP, CV, MA, MI, HL, and RL, with clear advantages under Non-IID data conditions. Statistical analyses confirm the significance of these gains. Future work will focus on stronger cryptographic techniques, lower computational costs, and extending the framework to vertical federated MFS and dynamic client participation.</p>
</sec>
</body>
<back>
<ack>
<p>During the preparation of this manuscript, the authors also utilized AI tools (e.g., ChatGPT 4.1 and ChatGPT 5) for language polishing and structural adjustments. The authors have carefully reviewed and revised the output and accept full responsibility for all content.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>This work was financially supported by the National Natural Science Foundation of China (Grants U23A20318, 62376089, 62302153, and 62302154), the Key Research and Development Program of Hubei Province (Grant 2023BEB024), and the Scientific and Technological Innovation Team Program for Young and Middle-Aged Researchers in Hubei Higher Education Institutions (Grant T2023007).</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>The authors&#x2019; contributions to the paper are as follows: Study conception and design: Songsong Zhang, Huazhong Jin, Zhiwei Ye; Data collection: Huazhong Jin, Zhiwei Ye, Songsong Zhang, Jia Yang, Jixin Zhang; Analysis and interpretation of results: Zhiwei Ye, Songsong Zhang, Xiao Zheng, Dongfang Wu; Draft manuscript preparation: Songsong Zhang, Dingfeng Song. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>This study uses datasets from two sources: the Mulan repository (<ext-link ext-link-type="uri" xlink:href="https://mulan.sourceforge.net/datasets-mlc.html">https://mulan.sourceforge.net/datasets-mlc.html</ext-link>) (accessed on 20 August 2025) and a multi-label classification database (<ext-link ext-link-type="uri" xlink:href="http://www.uco.es/kdis/mllresources/">http://www.uco.es/kdis/mllresources/</ext-link>) (accessed on 20 August 2025). For additional information or data requests, please contact the corresponding author.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>W</given-names></string-name>, <string-name><surname>Ding</surname> <given-names>W</given-names></string-name>, <string-name><surname>Li</surname> <given-names>T</given-names></string-name></person-group>. <article-title>Fusion-enhanced multi-label feature selection with sparse supplementation</article-title>. <source>Inf Fusion</source>. <year>2025</year>;<volume>117</volume>(<issue>3</issue>):<fpage>102813</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.inffus.2024.102813</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Fan</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Du</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Learning correlation information for multi-label feature selection</article-title>. <source>Pattern Recognit</source>. <year>2024</year>;<volume>145</volume>(<issue>8</issue>):<fpage>109899</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.patcog.2023.109899</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Sharma</surname> <given-names>M</given-names></string-name>, <string-name><surname>Tomar</surname> <given-names>A</given-names></string-name>, <string-name><surname>Hazra</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Edge computing for industry 5.0: fundamental, applications and research challenges</article-title>. <source>IEEE Internet Things J</source>. <year>2024</year>;<volume>11</volume>(<issue>11</issue>):<fpage>19070</fpage>&#x2013;<lpage>93</lpage>. doi:<pub-id pub-id-type="doi">10.1109/jiot.2024.3359297</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>H</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>M</given-names></string-name>, <string-name><surname>Xiong</surname> <given-names>H</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>S</given-names></string-name></person-group>. <article-title>When federated learning meets privacy-preserving computation</article-title>. <source>ACM Comput Surv</source>. <year>2024</year>;<volume>56</volume>(<issue>12</issue>):<fpage>1</fpage>&#x2013;<lpage>36</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3679013</pub-id>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tarekegn</surname> <given-names>AN</given-names></string-name>, <string-name><surname>Giacobini</surname> <given-names>M</given-names></string-name>, <string-name><surname>Michalak</surname> <given-names>K</given-names></string-name></person-group>. <article-title>A review of methods for imbalanced multi-label classification</article-title>. <source>Pattern Recognit</source>. <year>2021</year>;<volume>118</volume>:<fpage>107965</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.patcog.2021.107965</pub-id>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Maurya</surname> <given-names>SK</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Murata</surname> <given-names>T</given-names></string-name></person-group>. <article-title>Feature selection: key to enhance node classification with graph neural networks</article-title>. <source>CAAI Trans Intell Technol</source>. <year>2023</year>;<volume>8</volume>(<issue>1</issue>):<fpage>14</fpage>&#x2013;<lpage>28</lpage>. doi:<pub-id pub-id-type="doi">10.1049/cit2.12166</pub-id>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>X</given-names></string-name></person-group>. <article-title>Multi-label feature selection based on logistic regression and manifold learning</article-title>. <source>Appl Intell</source>. <year>2022</year>;<volume>52</volume>(<issue>8</issue>):<fpage>9256</fpage>&#x2013;<lpage>73</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10489-021-03008-8</pub-id>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Guo</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>T</given-names></string-name>, <string-name><surname>Li</surname> <given-names>YJ</given-names></string-name>, <string-name><surname>Deng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Qian</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Semi-supervised feature selection based on fuzzy related family</article-title>. <source>Inf Sci</source>. <year>2024</year>;<volume>652</volume>(<issue>6</issue>):<fpage>119660</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.ins.2023.119660</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Pandithurai</surname> <given-names>O</given-names></string-name>, <string-name><surname>Venkataiah</surname> <given-names>C</given-names></string-name>, <string-name><surname>Tiwari</surname> <given-names>S</given-names></string-name>, <string-name><surname>Ramanjaneyulu</surname> <given-names>N</given-names></string-name></person-group>. <article-title>DDoS attack prediction using a honey badger optimization algorithm based feature selection and Bi-LSTM in cloud environment</article-title>. <source>Expert Syst Appl</source>. <year>2024</year>;<volume>241</volume>(<issue>1</issue>):<fpage>122544</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.eswa.2023.122544</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Qian</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Hou</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Incremental feature spaces learning with label scarcity</article-title>. <source>ACM Trans Knowl Discov Data (TKDD)</source>. <year>2022</year>;<volume>16</volume>(<issue>6</issue>):<fpage>1</fpage>&#x2013;<lpage>26</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3516368</pub-id>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Fang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhai</surname> <given-names>H</given-names></string-name></person-group>. <article-title>A feature selection based on genetic algorithm for intrusion detection of industrial control systems</article-title>. <source>Comput Secur</source>. <year>2024</year>;<volume>139</volume>:<fpage>103675</fpage>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Karimi</surname> <given-names>F</given-names></string-name>, <string-name><surname>Dowlatshahi</surname> <given-names>MB</given-names></string-name>, <string-name><surname>Hashemi</surname> <given-names>A</given-names></string-name></person-group>. <article-title>SemiACO: a semi-supervised feature selection based on ant colony optimization</article-title>. <source>Expert Syst Appl</source>. <year>2023</year>;<volume>214</volume>(<issue>6</issue>):<fpage>119130</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.eswa.2022.119130</pub-id>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mei</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Ye</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>W</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A cooperative hybrid breeding swarm intelligence algorithm for feature selection</article-title>. <source>Pattern Recognit</source>. <year>2025</year>;<volume>169</volume>(<issue>12</issue>):<fpage>111901</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.patcog.2025.111901</pub-id>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ye</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>L</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>H</given-names></string-name></person-group>. <article-title>A hybrid rice optimization algorithm</article-title>. In: <conf-name>2016 11th International Conference on Computer Science &#x0026; Education (ICCSE); 2016 Aug 23&#x2013;25</conf-name>; <publisher-loc>Nagoya, Japan</publisher-loc>: <publisher-name>IEEE</publisher-name>. p. <fpage>169</fpage>&#x2013;<lpage>74</lpage>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cai</surname> <given-names>T</given-names></string-name>, <string-name><surname>Ye</surname> <given-names>C</given-names></string-name>, <string-name><surname>Ye</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Mei</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Multi-label feature selection based on improved ant colony optimization algorithm with dynamic redundancy and label dependence</article-title>. <source>Comput Mater Contin</source>. <year>2024</year>;<volume>81</volume>(<issue>1</issue>):<fpage>1157</fpage>&#x2013;<lpage>75</lpage>. doi:<pub-id pub-id-type="doi">10.32604/cmc.2024.055080</pub-id>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Hong</surname> <given-names>L</given-names></string-name>, <string-name><surname>Fu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Cai</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhong</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Adaptive distance-based multi-objective particle swarm optimization algorithm with simple position update</article-title>. <source>Swarm Evol Comput</source>. <year>2025</year>;<volume>94</volume>(<issue>5</issue>):<fpage>101890</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.swevo.2025.101890</pub-id>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhong</surname> <given-names>R</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Al-Shourbaji</surname> <given-names>I</given-names></string-name>, <string-name><surname>Houssein</surname> <given-names>EH</given-names></string-name>, <string-name><surname>Kachare</surname> <given-names>PH</given-names></string-name>, <string-name><surname>Jabbari</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Self-adaptive competitive swarm optimizer: a memetic approach for global optimization and human-powered aircraft design</article-title>. <source>Memetic Comput</source>. <year>2025</year>;<volume>17</volume>(<issue>3</issue>):<fpage>32</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s12293-025-00465-3</pub-id>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Mahanipour</surname> <given-names>A</given-names></string-name>, <string-name><surname>Khamfroush</surname> <given-names>H</given-names></string-name></person-group>. <article-title>FMLFS: a federated multi-label feature selection based on information theory in IoT environment</article-title>. In: <conf-name>2024 IEEE International Conference on Smart Computing (SMARTCOMP); 2024 Jun 29&#x2013;Jul 2</conf-name>; <publisher-loc>Osaka, Japan</publisher-loc>: <publisher-name>IEEE</publisher-name>. p. <fpage>166</fpage>&#x2013;<lpage>73</lpage>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Mahanipour</surname> <given-names>A</given-names></string-name>, <string-name><surname>Khamfroush</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Fuzzy federated multi-label feature selection: reinforcement learning and ant colony optimization</article-title>. In: <conf-name>2024 IEEE International Conference on Big Data (BigData); 2024 Dec 15&#x2013;18</conf-name>; <publisher-loc>Washington, DC, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>. p. <fpage>7919</fpage>&#x2013;<lpage>28</lpage>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Song</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>D</given-names></string-name>, <string-name><surname>Miao</surname> <given-names>J</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>K</given-names></string-name></person-group>. <article-title>Causal multi-label feature selection in federated setting</article-title>. <comment>arXiv:2403.06419. 2024</comment>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cheng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>W</given-names></string-name>, <string-name><surname>Qin</surname> <given-names>S</given-names></string-name>, <string-name><surname>Tu</surname> <given-names>T</given-names></string-name></person-group>. <article-title>Differential privacy federated learning based on adaptive adjustment</article-title>. <source>Comput Mater Contin</source>. <year>2025</year>;<volume>82</volume>(<issue>3</issue>):<fpage>4777</fpage>&#x2013;<lpage>95</lpage>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Ekanut</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>Augmented multi-party computation against gradient leakage in federated learning</article-title>. <source>IEEE Trans Big Data</source>. <year>2024</year>;<volume>10</volume>(<issue>6</issue>):<fpage>742</fpage>&#x2013;<lpage>51</lpage>. doi:<pub-id pub-id-type="doi">10.1109/tbdata.2022.3208736</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gao</surname> <given-names>C</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>J</given-names></string-name>, <string-name><surname>Miao</surname> <given-names>D</given-names></string-name>, <string-name><surname>Yue</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Granular-conditional-entropy-based attribute reduction for partially labeled data with proxy labels</article-title>. <source>Inf Sci</source>. <year>2021</year>;<volume>580</volume>(<issue>7</issue>):<fpage>111</fpage>&#x2013;<lpage>28</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.ins.2021.08.067</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dai</surname> <given-names>J</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>W</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Multi-label feature selection based on fuzzy mutual information and orthogonal regression</article-title>. <source>IEEE Trans Fuzzy Syst</source>. <year>2024</year>;<volume>32</volume>(<issue>9</issue>):<fpage>5136</fpage>&#x2013;<lpage>48</lpage>. doi:<pub-id pub-id-type="doi">10.1109/tfuzz.2024.3415176</pub-id>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Sparse multi-label feature selection via pseudo-label learning and dynamic graph constraints</article-title>. <source>Inf Fusion</source>. <year>2025</year>;<volume>118</volume>(<issue>5</issue>):<fpage>102975</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.inffus.2025.102975</pub-id>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Huang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Li</surname> <given-names>G</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>X</given-names></string-name></person-group>. <article-title>Learning label-specific features and class-dependent labels for multi-label classification</article-title>. <source>IEEE Trans Knowl Data Eng</source>. <year>2016</year>;<volume>28</volume>(<issue>12</issue>):<fpage>3309</fpage>&#x2013;<lpage>23</lpage>. doi:<pub-id pub-id-type="doi">10.1109/tkde.2016.2608339</pub-id>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Jian</surname> <given-names>L</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Shu</surname> <given-names>K</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Multi-label informed feature selection</article-title>. In: <conf-name>Proceedings of the Twenty-Fifth International Joint Conference on Artificial Intelligence (IJCAI-16); 2019 Jul 9&#x2013;15</conf-name>; <publisher-loc>New York, NY, USA</publisher-loc>. p. <fpage>1627</fpage>&#x2013;<lpage>33</lpage>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Sun</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Partial multi-label feature selection via low-rank and sparse factorization with manifold learning</article-title>. <source>Knowl Based Syst</source>. <year>2024</year>;<volume>296</volume>(<issue>8</issue>):<fpage>111899</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.knosys.2024.111899</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>C</given-names></string-name>, <string-name><surname>Li</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Manifold regularized discriminative feature selection for multi-label learning</article-title>. <source>Pattern Recognit</source>. <year>2019</year>;<volume>95</volume>(<issue>9</issue>):<fpage>136</fpage>&#x2013;<lpage>50</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.patcog.2019.06.003</pub-id>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Huo</surname> <given-names>W</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Multi-label feature selection via latent representation learning and dynamic graph constraints</article-title>. <source>Pattern Recognit</source>. <year>2024</year>;<volume>151</volume>(<issue>2</issue>):<fpage>110411</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.patcog.2024.110411</pub-id>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Mahanipour</surname> <given-names>A</given-names></string-name>, <string-name><surname>Khamfroush</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Wrapper-based federated feature selection for IoT environments</article-title>. In: <conf-name>2023 International Conference on Computing, Networking and Communications (ICNC); 2023 Feb 20&#x2013;22</conf-name>; <publisher-loc>Honolulu, HI, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>. p. <fpage>214</fpage>&#x2013;<lpage>9</lpage>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Feng</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Vertical federated learning-based feature selection with non-overlapping sample utilization</article-title>. <source>Expert Syst Appl</source>. <year>2022</year>;<volume>208</volume>(<issue>4</issue>):<fpage>118097</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.eswa.2022.118097</pub-id>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Cai</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Heidari</surname> <given-names>AA</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Weighted mean of vectors algorithm with neighborhood information interaction and vertical and horizontal crossover mechanism for feature selection</article-title>. <source>Appl Intell</source>. <year>2025</year>;<volume>55</volume>(<issue>1</issue>):<fpage>1</fpage>&#x2013;<lpage>44</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10489-024-05889-x</pub-id>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>de Greeff</surname> <given-names>J</given-names></string-name>, <string-name><surname>de Boer</surname> <given-names>MHT</given-names></string-name>, <string-name><surname>Hillerstr&#x00F6;m</surname> <given-names>FHJ</given-names></string-name>, <string-name><surname>Bomhof</surname> <given-names>H</given-names></string-name>, <string-name><surname>Jorritsma</surname> <given-names>W</given-names></string-name>, <string-name><surname>Neerincx</surname> <given-names>M</given-names></string-name></person-group>. <article-title>The FATE system: fair, transparent and explainable decision making</article-title>. In: <conf-name>AAAI Spring Symposium: Combining Machine Learning with Knowledge Engineering; 2021 Mar 22&#x2013;24</conf-name>; <publisher-loc>Palo Alto, CA, USA</publisher-loc>. p. <fpage>266</fpage>&#x2013;<lpage>7</lpage>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Yao</surname> <given-names>X</given-names></string-name></person-group>. <article-title>MOEA/D with spatial-temporal topological tensor prediction for evolutionary dynamic multiobjective optimization</article-title>. <source>IEEE Trans Evol Comput</source>. <year>2025</year>;<volume>29</volume>(<issue>3</issue>):<fpage>764</fpage>&#x2013;<lpage>78</lpage>. doi:<pub-id pub-id-type="doi">10.1109/tevc.2024.3367747</pub-id>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Arya</surname> <given-names>A</given-names></string-name>, <string-name><surname>Gunarani</surname> <given-names>GI</given-names></string-name>, <string-name><surname>Rathinakumar</surname> <given-names>V</given-names></string-name>, <string-name><surname>Sharma</surname> <given-names>A</given-names></string-name>, <string-name><surname>Pati</surname> <given-names>AK</given-names></string-name>, <string-name><surname>Sethi</surname> <given-names>KC</given-names></string-name></person-group>. <article-title>NSGA-III based optimization model for balancing time, cost, and quality in resource-constrained retrofitting projects</article-title>. <source>Asian J Civil Eng</source>. <year>2024</year>;<volume>25</volume>(<issue>7</issue>):<fpage>5613</fpage>&#x2013;<lpage>25</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s42107-024-01133-6</pub-id>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>M</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Li</surname> <given-names>S</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Group-preserving label-specific feature selection for multi-label learning</article-title>. <source>Expert Syst Appl</source>. <year>2023</year>;<volume>213</volume>:<fpage>118861</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.eswa.2022.118861</pub-id>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Miao</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Cheng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Parallel dual-channel multi-label feature selection</article-title>. <source>Soft Comput</source>. <year>2023</year>;<volume>27</volume>(<issue>11</issue>):<fpage>7115</fpage>&#x2013;<lpage>30</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s00500-023-07916-4</pub-id>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kashef</surname> <given-names>S</given-names></string-name>, <string-name><surname>Nezamabadi-Pour</surname> <given-names>H</given-names></string-name>, <string-name><surname>Nikpour</surname> <given-names>B</given-names></string-name></person-group>. <article-title>Multilabel feature selection: a comprehensive review and guiding experiments</article-title>. <source>Wiley Interdiscip Rev Data Mining Knowl Discov</source>. <year>2018</year>;<volume>8</volume>(<issue>2</issue>):<fpage>e1240</fpage>. doi:<pub-id pub-id-type="doi">10.1002/widm.1240</pub-id>.</mixed-citation></ref>
</ref-list>
</back></article>