<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">67589</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2025.067589</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Leveraging Federated Learning for Efficient Privacy-Enhancing Violent Activity Recognition from Videos</article-title>
<alt-title alt-title-type="left-running-head">Leveraging Federated Learning for Efficient Privacy-Enhancing Violent Activity Recognition from Videos</alt-title>
<alt-title alt-title-type="right-running-head">Leveraging Federated Learning for Efficient Privacy-Enhancing Violent Activity Recognition from Videos</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0009-0004-3628-8242</contrib-id>
<name name-style="western"><surname>Tonmoy</surname><given-names>Moshiur Rahman</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0009-0001-4883-1802</contrib-id>
<name name-style="western"><surname>Hossain</surname><given-names>Md. Mithun</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-3" contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-7445-7121</contrib-id>
<name name-style="western"><surname>Safran</surname><given-names>Mejdl</given-names></name><xref ref-type="aff" rid="aff-2">2</xref><email>mejdl@ksu.edu.sa</email></contrib>
<contrib id="author-4" contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0009-0001-1268-9613</contrib-id>
<name name-style="western"><surname>Alfarhood</surname><given-names>Sultan</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-5" contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-9906-4295</contrib-id>
<name name-style="western"><surname>Che</surname><given-names>Dunren</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-6" contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0001-5738-1631</contrib-id>
<name name-style="western"><surname>Mridha</surname><given-names>M. F.</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<aff id="aff-1"><label>1</label><institution>Department of Computer Science and Engineering, Bangladesh University of Business and Technology</institution>, <addr-line>Dhaka, 1216</addr-line>, <country>Bangladesh</country></aff>
<aff id="aff-2"><label>2</label><institution>Research Chair of Online Dialogue and Cultural Communication, Department of Computer Science, College of Computer and Information Sciences, King Saud University</institution>, <addr-line>Riyadh, 12372</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-3"><label>3</label><institution>Department of Electrical Engineering and Computer Science, Texas A&#x0026;M University-Kingsville</institution>, <addr-line>Kingsville, TX 78363</addr-line>, <country>USA</country></aff>
<aff id="aff-4"><label>4</label><institution>Department of Computer Science, American International University-Bangladesh</institution>, <addr-line>Dhaka, 1229</addr-line>, <country>Bangladesh</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Mejdl Safran. Email: <email>mejdl@ksu.edu.sa</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2025</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>23</day><month>10</month><year>2025</year>
</pub-date>
<volume>85</volume>
<issue>3</issue>
<fpage>5747</fpage>
<lpage>5763</lpage>
<history>
<date date-type="received">
<day>07</day>
<month>5</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>9</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 The Authors.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_67589.pdf"></self-uri>
<abstract>
<p>Automated recognition of violent activities from videos is vital for public safety, but often raises significant privacy concerns due to the sensitive nature of the footage. Moreover, resource constraints often hinder the deployment of deep learning-based complex video classification models on edge devices. With this motivation, this study aims to investigate an effective violent activity classifier while minimizing computational complexity, attaining competitive performance, and mitigating user data privacy concerns. We present a lightweight deep learning architecture with fewer parameters for efficient violent activity recognition. We utilize a two-stream formation of 3D depthwise separable convolution coupled with a linear self-attention mechanism for effective feature extraction, incorporating federated learning to address data privacy concerns. Experimental findings demonstrate the model&#x2019;s effectiveness with test accuracies from 96% to above 97% on multiple datasets by incorporating the FedProx aggregation strategy. These findings underscore the potential to develop secure, efficient, and reliable solutions for violent activity recognition in real-world scenarios.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Violent activity recognition</kwd>
<kwd>human activity recognition</kwd>
<kwd>federated learning</kwd>
<kwd>video understanding</kwd>
<kwd>computer vision</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>King Saud University</funding-source>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Human activity recognition (HAR) systems and video surveillance have been significantly enhanced by artificial intelligence (AI), enabling advancements in crowd analysis, behavior monitoring, and public safety [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-2">2</xref>]. These AI-driven developments facilitate the automated recognition of violent activities, which is crucial for maintaining security in public spaces. For instance, recent data indicate that the global video surveillance market is expected to reach $88.71 billion by 2030, with AI integration being the key driver of this growth [<xref ref-type="bibr" rid="ref-3">3</xref>]. However, the integration of AI in surveillance raises critical concerns regarding data security, privacy, and substantial processing demands associated with high-resolution video data [<xref ref-type="bibr" rid="ref-4">4</xref>]. Violence recognition systems often analyze sensitive and identifiable information, thereby increasing the risk of privacy infringement [<xref ref-type="bibr" rid="ref-5">5</xref>]. In response to stringent data protection regulations such as the California Consumer Privacy Act (CCPA) and the General Data Protection Regulation (GDPR), there is a growing imperative for AI systems that not only deliver reliable performance but also provide robust privacy safeguards [<xref ref-type="bibr" rid="ref-6">6</xref>,<xref ref-type="bibr" rid="ref-7">7</xref>].</p>
<p>Addressing these challenges requires innovative approaches to ensure the privacy and efficiency of AI-based automated video surveillance systems. Surveillance videos frequently contain sensitive information, which necessitates stringent privacy measures. In addition, the analysis of large-scale video data poses significant memory and processing challenges, particularly when leveraging cloud-based servers, which can lead to accessibility issues for edge devices [<xref ref-type="bibr" rid="ref-8">8</xref>]. To mitigate these obstacles, researchers have explored privacy-preserving techniques such as secure multiparty computation, federated learning (FL), and differential privacy [<xref ref-type="bibr" rid="ref-5">5</xref>]. Among these, FL has gained prominence for enabling decentralized learning without compromising data privacy. There are several paradigms and variants of FL based on different criteria. For instance, the most common is the horizontal FL, where clients share the same feature space but differ in samples (e.g., surveillance cameras at different locations training on similar video features). In vertical FL, clients share samples but differ in features (e.g., different sensors for the same event), and federated transfer learning combines both horizontal and vertical settings when clients differ in both samples and features [<xref ref-type="bibr" rid="ref-9">9</xref>,<xref ref-type="bibr" rid="ref-10">10</xref>]. Despite these advancements, many existing methods encounter limitations related to real-time scalability, high memory consumption owing to encryption protocols, and excessive computational complexity [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-12">12</xref>]. Furthermore, achieving a balance between high classification accuracy and robust privacy guarantees remains challenging, particularly in resource-constrained environments or scenarios with noisy data [<xref ref-type="bibr" rid="ref-13">13</xref>]. The high memory demands of deep and complex models create barriers to their deployment on resource-constrained edge devices, emphasizing the need for lightweight deep learning (DL) architectures [<xref ref-type="bibr" rid="ref-14">14</xref>].</p>
<p>Motivated by the aforementioned challenges, we investigate an FL-based framework for privacy-enhancing decentralized learning and efficient violence recognition from videos. We incorporate a horizontal FL approach since each surveillance camera or client would hold the same feature space, i.e., video frames with the same modalities but different samples. Our DL classifier emphasizes a lightweight design, recognizing that both FL and on-device deployment shift the computational burden of the entire network to local devices that typically lack high-end processing capabilities. To achieve competitive accuracy, our model employs a two-stream architecture that integrates multiple blocks of 3D depthwise separable convolutions with a linear self-attention mechanism, which enables efficient extraction of spatial and temporal features. In addition, we enforced information fusion between streams through residual connections, thereby enhancing information flow and convergence. The key aspects of this study can be summarized as follows:
<list list-type="bullet">
<list-item>
<p>We propose an innovative and effective framework for violent activity recognition integrated with FL to enhance user data privacy through decentralized learning</p></list-item>
<list-item>
<p>Our lightweight model contains only approximately 1.104 million parameters and 4.42 MB in size, significantly fewer than state-of-the-art video classifiers, which typically require approximately 100 times more parameters</p></list-item>
<list-item>
<p>Experiments on multiple datasets demonstrated competitive recognition accuracies attained by the model</p></list-item>
<list-item>
<p>Proposed model outperformed popular models such as ViViT and TimeSformer in a comprehensive performance analysis.</p></list-item>
</list></p>
<p>The rest of the paper is organized as follows: <xref ref-type="sec" rid="s2">Section 2</xref> presents an overview of past works. <xref ref-type="sec" rid="s3">Section 3</xref> discusses the methodology in detail. <xref ref-type="sec" rid="s4">Section 4</xref> summarizes the experiment and findings, followed by a discussion of the overall study in <xref ref-type="sec" rid="s5">Section 5</xref>. Finally, <xref ref-type="sec" rid="s6">Section 6</xref> concludes the study.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Works</title>
<p>Recent advancements in violence recognition within surveillance videos have utilized various DL architectures to enhance accuracy and real-time processing. Ullah et al. [<xref ref-type="bibr" rid="ref-15">15</xref>] conducted a comprehensive review of ML and DL methods for violence detection, highlighting existing challenges and future directions focused on surveillance scenarios. Liu et al. [<xref ref-type="bibr" rid="ref-16">16</xref>] proposed a human-centered attention mechanism that can dynamically emphasize the salient regions from videos associated with the action, leading to effective recognition. Aggarwal et al. [<xref ref-type="bibr" rid="ref-17">17</xref>] introduced a system combining MobileNetV2 with a BiLSTM layer, achieving 96% accuracy on CCTV footage. Similarly, Yadav et al. [<xref ref-type="bibr" rid="ref-18">18</xref>] presented a CNN and LSTM framework that achieved up to 98% accuracy and a processing speed of 131 frames/sec. Abbass and Kang [<xref ref-type="bibr" rid="ref-19">19</xref>] enhanced detection by integrating Convolutional Block Attention Modules (CBAM) and using data augmentation and Categorical Focal Loss to address class imbalance. Kumar et al. [<xref ref-type="bibr" rid="ref-20">20</xref>] proposed a lightweight transformer model for indoor violence detection that achieved up to 98% accuracy with occluded subjects. Mohammadi and Nazerfard [<xref ref-type="bibr" rid="ref-21">21</xref>] introduced a semi-supervised hard attention model using reinforcement learning, achieving high accuracies on the RWF and Hockey datasets, although it depends heavily on the quality of the reinforcement learning algorithms and may underperform in less controlled settings. Finally, Vijeikis et al. [<xref ref-type="bibr" rid="ref-22">22</xref>] combined a U-Net-like network with MobileNet V2 and an LSTM module, attaining around 82% accuracy and 81% precision on the RWF-2000 dataset. Many frameworks for protecting sensitive data have also been made available by recent advances in privacy-preserving video analytics. For instance, Frimpong et al. [<xref ref-type="bibr" rid="ref-23">23</xref>] developed Secrets In Motion (SIM), which employs Ciphertext Policy Attribute-Based Encryption (CP-ABE) and Multi-Key Homomorphic Encryption (MKHE) to control access to video content. Although SIM balances privacy with classification accuracy, it may face scalability issues and the computational complexity of its encryption methods, limiting its use in large-scale deployments. Feng et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] introduced X-Stream, a flexible video transformer for privacy-preserving video stream analytics, featuring a declarative query interface for specifying privacy and content exposure preferences, an adaptive mechanism that selects appropriate privacy-preserving techniques at runtime, and an efficient execution engine optimized for multi-task deduplication and inter-frame inference. Gaikwad and Karmakar [<xref ref-type="bibr" rid="ref-25">25</xref>] presented the Privacy-Aware Person Search (PAPS) model for IoT surveillance, which processes data at the edge and fog layers to minimize privacy risks. Mehta et al. [<xref ref-type="bibr" rid="ref-26">26</xref>] introduced SETR-PKD, a seizure detection framework that uses optical flow features to preserve patient confidentiality. However, this method can be less effective in environments with minimal movement or where motion patterns are subtle, reducing its reliability in diverse settings. Singh et al. [<xref ref-type="bibr" rid="ref-27">27</xref>] employed ViViT for violence detection, incorporating data augmentation to enhance performance on smaller datasets. On the other hand, Pajon et al. [<xref ref-type="bibr" rid="ref-28">28</xref>] modified the Flow-Gated model and proposed an innovative approach called &#x201C;Diff Gated&#x201D; network, which attained improved results compared to the original model. However, they only experimented with the federated averaging (FedAvg) aggregation strategy, which assumes that data across clients follows a similar distribution, which is often unrealistic in surveillance settings. Similarly, Victor et al. [<xref ref-type="bibr" rid="ref-29">29</xref>] proposed an FL&#x2013;based violence detection framework that extracts individual frames from the AIRTLab dataset and classifies them using pre-trained CNN backbones. However, their approach is limited to frame-level modeling, thus overlooking critical temporal dynamics, and is evaluated on a single dataset, raising concerns about the generalizability.</p>
<p>Earlier studies have primarily concentrated on enhancing human activity recognition (HAR) methods to improve the efficiency of recognition models. However, these advancements often come with significant computational demands and fail to adequately address crucial factors such as user data privacy, especially when dealing with sensitive footage of daily life or indoor activities. While some studies have explored privacy concerns such as incorporating FL, their approaches are often limited by issues such as a lack of convincing performance, poor generalizability, and high computational complexity that hinders scalability. Additionally, many existing method poses the risk of high false positive rates and inconsistent performance across diverse real-world settings. Therefore, it is essential to develop lightweight architectures that effectively balance accuracy and model complexity. Overcoming these challenges is critical for creating scalable, secure, and efficient solutions for practical applications.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Methodology</title>
<p>In this section, we first provide an overview of Federated Learning, followed by the introduction of our proposed DL-based architecture for efficient and effective violent activity recognition, leveraging Federated Learning.</p>
<sec id="s3_1">
<label>3.1</label>
<title>Federated Learning</title>
<p>Federated Learning, also known as FL, contrasts with traditional centralized learning by enabling multiple entities (known as clients) to train a model collaboratively without sharing private data [<xref ref-type="bibr" rid="ref-30">30</xref>]. Rather, clients only share model updates (e.g., weights or gradients) with the central server. The server aggregates these local updates to construct a global model and distributes the updated global parameters to the entities. The design of FL enhances data privacy and has broad applications across domains such as healthcare, surveillance, and beyond [<xref ref-type="bibr" rid="ref-31">31</xref>,<xref ref-type="bibr" rid="ref-32">32</xref>]. Aggregation strategies vary according to the target application, data distribution, and other factors. An efficient aggregation strategy is one of the popular research domains in FL, and researchers have explored various effective strategies for tackling various challenges while improving performance. In this study, we experimented with two popular strategies that are briefly discussed in the following subsections.</p>
<sec id="s3_1_1">
<label>3.1.1</label>
<title>Federated Averaging (FedAvg)</title>
<p>Federated Averaging, widely known as FedAvg, is the most common and intuitive aggregation strategy for FL [<xref ref-type="bibr" rid="ref-30">30</xref>]. It aggregates model updates from multiple clients by simply calculating the weighted average of their locally computed updates, weighted by the size of each client&#x2019;s dataset. Mathematically,
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:msub><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover><mml:mfrac><mml:msub><mml:mi>N</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mi>N</mml:mi></mml:mfrac><mml:msubsup><mml:mi>w</mml:mi><mml:mi>t</mml:mi><mml:mi>k</mml:mi></mml:msubsup></mml:math></disp-formula>where, <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:msub><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> is the aggregated global model weights at round <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mi>t</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:msubsup><mml:mi>w</mml:mi><mml:mi>t</mml:mi><mml:mi>k</mml:mi></mml:msubsup></mml:math></inline-formula> is the update from the <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mi>k</mml:mi></mml:math></inline-formula>-th client at round <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mi>t</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:msub><mml:mi>N</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:math></inline-formula> is the number of data samples at the <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mi>k</mml:mi></mml:math></inline-formula>-th client, <italic>N</italic> is total data samples across all clients, and <italic>K</italic> is the total number of participating clients.</p>
</sec>
<sec id="s3_1_2">
<label>3.1.2</label>
<title>Federated Proximal (FedProx)</title>
<p>The FedProx algorithm is an extension of FedAvg designed to address the heterogeneity of client data distributions and system constraints in FL [<xref ref-type="bibr" rid="ref-33">33</xref>]. It introduces a proximal term in the loss function that penalizes significant deviations between the local client model parameters and the global model parameters. The proximal term is scaled by the parameter <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mi>&#x03BC;</mml:mi></mml:math></inline-formula>, which controls the strength of the penalty. The optimization objective of FedProx can be expressed as:
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:munder><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mfrac><mml:mi>&#x03BC;</mml:mi><mml:mn>2</mml:mn></mml:mfrac><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msup><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:msup><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>]</mml:mo></mml:mrow></mml:math></disp-formula>where, <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:msup><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:msup></mml:math></inline-formula> is the global parameters in round <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mi>t</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:msub><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:math></inline-formula> is the local parameters for client <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:mi>k</mml:mi></mml:math></inline-formula> after training, <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:msub><mml:mi>F</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is the local loss function for client <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mi>k</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:mi>&#x03BC;</mml:mi></mml:math></inline-formula> is the proximal term coefficient (<inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mi>&#x03BC;</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula>), and <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msup><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:msup><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula> is the proximal term.</p>
</sec>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Proposed Architecture</title>
<p>The framework incorporates decentralized training based on FL principles, operating across multiple clients or data sources. As illustrated in Algorithm 1, the process begins with the initialization of the global model <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:msub><mml:mi>M</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:math></inline-formula> on a central cloud server. The server then broadcasts an instance of <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:msub><mml:mi>M</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:math></inline-formula> to all participating clients <italic>C</italic> (i.e., edge devices), where each client trains the local model <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:msub><mml:mi>M</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula> using its private data. After training, each client sends the parameters of <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:msub><mml:mi>M</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula> back to the server. The server aggregates these updates using a predefined strategy (e.g., FedAvg, FedProx, etc.), updates <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:msub><mml:mi>M</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:math></inline-formula> with the aggregated parameters, and redistributes the updated <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:msub><mml:mi>M</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:math></inline-formula> to all clients for the next training round. This iterative process continues until the global model converges or until the maximum number of rounds <italic>R</italic> is reached. The FL mechanism ensures user data privacy by keeping raw data on local devices throughout the training process. Additionally, the lightweight design of the DL classifier enables the model to run entirely on edge devices without relying on cloud communications, therefore, On-Device AI-powered deployment will ensure that user data remains secure on the device, even when deployed for real-world recognition. A visual illustration of the framework is presented in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Overview of the FL empowered violent activity recognition</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67589-fig-1.tif"/>
</fig>
<fig id="fig-6">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67589-fig-6.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig-2">Fig. 2</xref> illustrates the proposed classifier architecture and its key components. <xref ref-type="fig" rid="fig-2">Fig. 2a</xref> provides an overview of the model, which comprises two-stream feature extraction, where each stream uses different filter sizes during the convolution operation to capture the diverse characteristics of the input frames. <xref ref-type="fig" rid="fig-2">Fig. 2b</xref> shows the building block of the two-stream formation, the DepthConv block, which employs 3D depthwise separable convolution. In this process, depthwise convolution treats each input channel independently, followed by pointwise convolution to combine the channels. This mechanism efficiently reduces the computational burden while extracting essential features [<xref ref-type="bibr" rid="ref-34">34</xref>]. Mathematically, given an input tensor <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:mi>D</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, first a 3D convolution is applied independently to each input channel:
<disp-formula id="ueqn-3"><mml:math id="mml-ueqn-3" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mi>d</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mi>d</mml:mi></mml:msub><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mi>d</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>K</mml:mi><mml:mi>d</mml:mi></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>K</mml:mi><mml:mi>h</mml:mi></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>K</mml:mi><mml:mi>w</mml:mi></mml:msub></mml:mrow></mml:msup></mml:math></disp-formula>where &#x2217; represents the convolution operation and <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mi>d</mml:mi></mml:msub></mml:math></inline-formula> is the depthwise kernel. Next, a <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> convolution is applied to project the feature maps to <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> channels:
<disp-formula id="ueqn-4"><mml:math id="mml-ueqn-4" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mi>p</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mi>p</mml:mi></mml:msub><mml:mo>&#x2217;</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mi>d</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">b</mml:mtext></mml:mrow><mml:mi>p</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mi>p</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></disp-formula>where <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mi>p</mml:mi></mml:msub></mml:math></inline-formula> is the pointwise kernel, and <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">b</mml:mtext></mml:mrow><mml:mi>p</mml:mi></mml:msub></mml:math></inline-formula> is the bias. We also employed batch normalization within the DepthConv block to enhance training efficiency further. Additionally, the GELU activation function is used throughout the architecture to introduce non-linearity, as GELU offers smoother and more stable gradient propagation compared to ReLU, making it a preferred choice for many DL tasks [<xref ref-type="bibr" rid="ref-35">35</xref>].</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Outline of employed classifier and its components: (<bold>a</bold>) the classifier with two-stream feature extraction, (<bold>b</bold>) DepthConv Block comprising Depthwise Separable 3D Convolution, (<bold>c</bold>) Attention Block employing Separable Self-Attention over the temporal dimension</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67589-fig-2.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig-2">Fig. 2c</xref> illustrates the attention block, which is a key component of our model that enhances performance while minimizing computational complexity. This block employs a separable temporal self-attention mechanism that is integrated with a residual connection. The MobileViTV2 image classification model inspires the self-attention mechanism [<xref ref-type="bibr" rid="ref-36">36</xref>], which operates with linear complexity, making it well-suited for mobile devices. In our implementation, we adapted the attention mechanism to use 3D convolution and GELU activation instead of the linear operations with ReLU non-linearity for improved performance in video-based tasks. Given an input tensor <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>D</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, where <italic>C</italic> is the embedding dimension, the input is first normalized using Layer Normalization. Next, a <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> convolution projects the normalized input into query (<inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:mrow><mml:mtext mathvariant="bold">Q</mml:mtext></mml:mrow></mml:math></inline-formula>), key (<inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:mrow><mml:mtext mathvariant="bold">K</mml:mtext></mml:mrow></mml:math></inline-formula>), and value (<inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:mrow><mml:mtext mathvariant="bold">V</mml:mtext></mml:mrow></mml:math></inline-formula>) tensors:
<disp-formula id="ueqn-5"><mml:math id="mml-ueqn-5" display="block"><mml:mrow><mml:mtext mathvariant="bold">Q</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mtext mathvariant="bold">K</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mtext mathvariant="bold">V</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mtext>split</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mrow><mml:mi>q</mml:mi><mml:mi>k</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2217;</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mi>n</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mrow><mml:mi>q</mml:mi><mml:mi>k</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mn>2</mml:mn><mml:mi>C</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x00D7;</mml:mo><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> is the convolutional kernel. The context scores (<italic>A</italic>) are then computed using softmax over the temporal dimension of the <italic>Q</italic>, and the <italic>K</italic> is weighted by the context scores and summed along the attention dimension for context vector (<italic>C</italic>):
<disp-formula id="ueqn-6"><mml:math id="mml-ueqn-6" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mrow><mml:mtext mathvariant="bold">A</mml:mtext></mml:mrow></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mtext>softmax</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">Q</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mtext mathvariant="bold">C</mml:mtext></mml:mrow></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mtext mathvariant="bold">A</mml:mtext></mml:mrow><mml:mo>&#x2299;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">K</mml:mtext></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:mo>&#x2299;</mml:mo></mml:math></inline-formula> denotes element-wise multiplication. The output is refined by applying GELU activation to the value tensor (<italic>V</italic>), then modulating it with the expanded context vector (<italic>C</italic>):
<disp-formula id="ueqn-7"><mml:math id="mml-ueqn-7" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">Y</mml:mtext></mml:mrow><mml:mi>a</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mtext>GELU</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">V</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2299;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">C</mml:mtext></mml:mrow></mml:math></disp-formula></p>
<p>A final <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> convolution projects back to the embedding dimension:
<disp-formula id="ueqn-8"><mml:math id="mml-ueqn-8" display="block"><mml:mrow><mml:mtext mathvariant="bold">Y</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mi>o</mml:mi></mml:msub><mml:mo>&#x2217;</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">Y</mml:mtext></mml:mrow><mml:mi>a</mml:mi></mml:msub></mml:math></disp-formula>where <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mi>o</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>. Lastly, dropout is applied, and a residual connection is added:
<disp-formula id="ueqn-9"><mml:math id="mml-ueqn-9" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>Dropout</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">Y</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<p>In the end, we employed global average pooling to summarize the overall influence of the extracted features for classification. The resulting tensor is flattened and passed through a linear layer with a dropout operation to prevent overfitting. The final classification was achieved using softmax scores.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experiment and Result</title>
<p>This section outlines the experimental details and findings of our study, starting with the data preparation and training setup, followed by performance analysis and ablation study, and concluding with the performance comparison with baseline models.</p>
<sec id="s4_1">
<label>4.1</label>
<title>Experimental Data, and Setup</title>
<p>We conducted experiments using three separate datasets to evaluate the performance of our proposed model, simulating decentralized learning. The first dataset is the Hockey Fight (HF) detection dataset [<xref ref-type="bibr" rid="ref-37">37</xref>], which consists of 1000 clips (500 fight and 500 non-fight) from the National Hockey League (NHL). Each clip contains 50 frames with a resolution of 720 <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 576 pixels. The second dataset is the Weapon Violence Dataset 2.0 (WVD) [<xref ref-type="bibr" rid="ref-38">38</xref>], a synthetic dataset containing 334 videos, which are divided into 60 instances of hot violence (representing violence with firearms), 54 instances of cold violence (representing violence with traditional weapons), and 54 instances labeled as no violence. Lastly, we used the Firearm Action Recognition dataset [<xref ref-type="bibr" rid="ref-39">39</xref>], consisting of 398 videos representing actions with no gun (118 videos), handguns (141 videos), and machine guns (139 videos).</p>
<p>To prepare the datasets for the experiment, we downsampled the resolution of the original clips to 224 <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 224 pixels and uniformly sampled 32 frames from each clip to ensure consistency. In the HF dataset, we replaced the original class names &#x201C;fight&#x201D; and &#x201C;non-fight&#x201D; with &#x201C;violent&#x201D; and &#x201C;non-violent.&#x201D; Additionally, we merged the &#x201C;cold violence&#x201D; and &#x201C;hot violence&#x201D; clips from the WVD dataset into a single &#x201C;violent&#x201D; class. The datasets were then split into three parts, maintaining a 60-20-20 ratio for training, validation, and test evaluation. Extensive on-the-fly augmentation was applied during training using the Albumentations library [<xref ref-type="bibr" rid="ref-40">40</xref>], as summarized in <xref ref-type="table" rid="table-1">Table 1</xref>. All three data splits were normalized with a mean of (0.485, 0.456, 0.406) and a standard deviation of (0.229, 0.224, 0.225).</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Overview of employed on-the-fly data augmentation strategies</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Attribute</th>
<th>Probability</th>
</tr>
</thead>
<tbody>
<tr>
<td>Horizontal flip</td>
<td>50%</td>
</tr>
<tr>
<td>Random Brightness/Contrast</td>
<td>50%</td>
</tr>
<tr>
<td>Shift, Scale, Rotate (limits: 0.05, 0.05, <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:msup><mml:mn>30</mml:mn><mml:mrow><mml:mo>&#x2218;</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula>)</td>
<td>50%</td>
</tr>
<tr>
<td>RGB Shift (shift limits: 15)</td>
<td>50%</td>
</tr>
<tr>
<td>Hue, Saturation, Value shift (limits: 20, 30, 20)</td>
<td>50%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The experimental implementations were carried out using Python and PyTorch to simulate the FL training system. The training data were partitioned into two subsets to simulate two participating local clients, each using their respective local data. Each client was trained on their local data for two consecutive epochs per round. To enhance the training process, we implemented callback strategies, including learning rate reduction and early stopping, based on the validation set performance of the global model after aggregation. If the validation accuracy showed no improvement for five consecutive rounds, the learning rate was reduced by 50% globally. Early stopping was triggered if no improvement in validation accuracy occurred over 30 consecutive rounds. All training and testing experiments were conducted in the Kaggle Notebook environment, utilizing two NVIDIA Tesla T4 GPUs with 29 GB of RAM. A summary of the experimental settings for the proposed model is provided in <xref ref-type="table" rid="table-2">Table 2</xref>.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Summary of the experimental settings</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th>Attribute</th>
<th>Value</th>
</tr>
</thead>
<tbody>
<tr>
<td>Frame shape</td>
<td><inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>224</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>224</mml:mn></mml:math></inline-formula></td>
</tr>
<tr>
<td>No. of frames</td>
<td>32</td>
</tr>
<tr>
<td>Batch size</td>
<td>8</td>
</tr>
<tr>
<td>Initial learning rate</td>
<td><inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mn>1</mml:mn><mml:mi>e</mml:mi><mml:mspace width="negativethinmathspace" /><mml:mo>&#x2212;</mml:mo><mml:mspace width="negativethinmathspace" /><mml:mn>3</mml:mn></mml:math></inline-formula></td>
</tr>
<tr>
<td>Minimum learning rate</td>
<td><inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:mn>1</mml:mn><mml:mi>e</mml:mi><mml:mspace width="negativethinmathspace" /><mml:mo>&#x2212;</mml:mo><mml:mspace width="negativethinmathspace" /><mml:mn>8</mml:mn></mml:math></inline-formula></td>
</tr>
<tr>
<td>Max communication round</td>
<td>500</td>
</tr>
<tr>
<td>Local epoch</td>
<td>2</td>
</tr>
<tr>
<td>LR reduction patience</td>
<td>5</td>
</tr>
<tr>
<td>Early stopping patience</td>
<td>30</td>
</tr>
<tr>
<td>Weight decay</td>
<td><inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:mn>1</mml:mn><mml:mi>e</mml:mi><mml:mspace width="negativethinmathspace" /><mml:mo>&#x2212;</mml:mo><mml:mspace width="negativethinmathspace" /><mml:mn>3</mml:mn></mml:math></inline-formula></td>
</tr>
<tr>
<td>Dropout rate</td>
<td>40%</td>
</tr>
<tr>
<td>Activation</td>
<td>GELU</td>
</tr>
<tr>
<td>Optimizer</td>
<td>Adam</td>
</tr>
<tr>
<td>Loss</td>
<td>Categorical crossentropy</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Performance Analysis</title>
<p>The model consistently outperformed across all datasets when trained with the FedProx strategy compared to FedAvg. As shown in <xref ref-type="table" rid="table-3">Table 3</xref>, FedProx achieved a test accuracy of 97.50% and an AUC of 0.9958 on the HF dataset, whereas FedAvg reached only 94.99%, marking a 2.58% drop in accuracy. A similar trend was observed on the WVD dataset, where FedProx achieved 97.06% accuracy and an AUC of 0.9616&#x2014;an improvement of nearly 10% over the 88.24% accuracy obtained using FedAvg. On the Firearm dataset, FedProx also led to an 8.46% increase in accuracy compared to FedAvg. It is important to note that the small number of test samples per class had a significant impact on the final accuracy. For example, as illustrated in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, a single misclassification out of 34 samples in WVD yielded an accuracy of 97.06%. Similarly, five misclassifications out of 200 samples in HF resulted in 97.50% accuracy, while three misclassified samples out of 80 in the Firearm dataset led to a 96.25% accuracy.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Classwise performance on experimental test sets across global aggregation strategies (Acc. &#x003D; Accuracy, P &#x003D; Precision, R &#x003D; Recall)</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Dataset</th>
<th>Strategy</th>
<th>Classes</th>
<th>Acc (%)</th>
<th>P (%)</th>
<th>R (%)</th>
<th>F1-score</th>
<th>AUC</th>
<th>Support</th>
</tr>
</thead>
<tbody>
<tr>
<td>HF</td>
<td>FedAvg</td>
<td>Non-violent</td>
<td>&#x2013;</td>
<td>94.79</td>
<td>94.79</td>
<td>0.9479</td>
<td>&#x2013;</td>
<td>96</td>
</tr>
<tr>
<td></td>
<td></td>
<td>Violent</td>
<td>&#x2013;</td>
<td>95.19</td>
<td>95.19</td>
<td>0.9519</td>
<td>&#x2013;</td>
<td>104</td>
</tr>
<tr>
<td></td>
<td></td>
<td><bold>Overall</bold></td>
<td>94.99</td>
<td>94.99</td>
<td>94.99</td>
<td>0.9499</td>
<td>0.9874</td>
<td>200</td>
</tr>
<tr>
<td></td>
<td>FedProx</td>
<td>Non-violent</td>
<td>&#x2013;</td>
<td>97.90</td>
<td>96.88</td>
<td>0.9738</td>
<td>&#x2013;</td>
<td>96</td>
</tr>
<tr>
<td></td>
<td></td>
<td>Violent</td>
<td>&#x2013;</td>
<td>97.14</td>
<td>98.08</td>
<td>0.9761</td>
<td>&#x2013;</td>
<td>104</td>
</tr>
<tr>
<td></td>
<td></td>
<td><bold>Overall</bold></td>
<td>97.50</td>
<td>97.50</td>
<td>97.50</td>
<td>0.9750</td>
<td>0.9958</td>
<td>200</td>
</tr>
<tr>
<td>WVD</td>
<td>FedAvg</td>
<td>Non-violent</td>
<td>&#x2013;</td>
<td>75.00</td>
<td>75.00</td>
<td>0.7500</td>
<td>&#x2013;</td>
<td>8</td>
</tr>
<tr>
<td></td>
<td></td>
<td>Violent</td>
<td>&#x2013;</td>
<td>92.31</td>
<td>92.31</td>
<td>0.9231</td>
<td>&#x2013;</td>
<td>26</td>
</tr>
<tr>
<td></td>
<td></td>
<td><bold>Overall</bold></td>
<td>88.24</td>
<td>88.24</td>
<td>88.24</td>
<td>0.8824</td>
<td>0.8654</td>
<td>34</td>
</tr>
<tr>
<td></td>
<td>FedProx</td>
<td>Non-violent</td>
<td>&#x2013;</td>
<td>100.00</td>
<td>87.50</td>
<td>0.9333</td>
<td>&#x2013;</td>
<td>8</td>
</tr>
<tr>
<td></td>
<td></td>
<td>Violent</td>
<td>&#x2013;</td>
<td>96.30</td>
<td>100.00</td>
<td>0.9811</td>
<td>&#x2013;</td>
<td>26</td>
</tr>
<tr>
<td></td>
<td></td>
<td><bold>Overall</bold></td>
<td>97.06</td>
<td>97.17</td>
<td>97.06</td>
<td>0.9699</td>
<td>0.9616</td>
<td>34</td>
</tr>
<tr>
<td>Firearm</td>
<td>FedAvg</td>
<td>machine_gun</td>
<td>&#x2013;</td>
<td>86.96</td>
<td>76.92</td>
<td>0.8163</td>
<td>&#x2013;</td>
<td>26</td>
</tr>
<tr>
<td></td>
<td></td>
<td>Handgun</td>
<td>&#x2013;</td>
<td>80.65</td>
<td>100.00</td>
<td>0.8929</td>
<td>&#x2013;</td>
<td>25</td>
</tr>
<tr>
<td></td>
<td></td>
<td>no_gun</td>
<td>&#x2013;</td>
<td>100.00</td>
<td>89.66</td>
<td>0.9456</td>
<td>&#x2013;</td>
<td>29</td>
</tr>
<tr>
<td></td>
<td></td>
<td><bold>Overall</bold></td>
<td>88.75</td>
<td>89.71</td>
<td>88.75</td>
<td>0.8871</td>
<td>0.9802</td>
<td>80</td>
</tr>
<tr>
<td></td>
<td>FedProx</td>
<td>machine_gun</td>
<td>&#x2013;</td>
<td>96.00</td>
<td>92.31</td>
<td>0.9412</td>
<td>&#x2013;</td>
<td>26</td>
</tr>
<tr>
<td></td>
<td></td>
<td>Handgun</td>
<td>&#x2013;</td>
<td>92.31</td>
<td>96.00</td>
<td>0.9412</td>
<td>&#x2013;</td>
<td>25</td>
</tr>
<tr>
<td></td>
<td></td>
<td>no_gun</td>
<td>&#x2013;</td>
<td>100.00</td>
<td>100.00</td>
<td>1.0000</td>
<td>&#x2013;</td>
<td>29</td>
</tr>
<tr>
<td></td>
<td></td>
<td><bold>Overall</bold></td>
<td>96.25</td>
<td>96.30</td>
<td>96.25</td>
<td>0.9625</td>
<td>0.9952</td>
<td>80</td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Confusion matrix of the test set evaluation using the FedProx strategy. (<bold>a</bold>) HF dataset; (<bold>b</bold>) WVD dataset; (<bold>c</bold>) Firearm dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67589-fig-3.tif"/>
</fig>
<p>To analyze the impact of FedAvg and FedProx on performance, <xref ref-type="fig" rid="fig-4">Fig. 4</xref> illustrates the training accuracy and loss across the communication rounds for each client using the HF dataset. <xref ref-type="fig" rid="fig-4">Fig. 4a</xref>,<xref ref-type="fig" rid="fig-4">c</xref> shows that FedAvg completed only 43 rounds before triggering early stopping, while <xref ref-type="fig" rid="fig-4">Fig. 4b</xref>,<xref ref-type="fig" rid="fig-4">d</xref> demonstrates that FedProx sustained training for 78 rounds, progressively enhancing robustness, delaying the early stopping threshold, and improving global validation accuracy. <xref ref-type="fig" rid="fig-4">Fig. 4</xref> also illustrates the average training accuracy and loss with the global validation performance. Although FedAvg initially achieved a peak validation accuracy of 95.50%, its performance deteriorated in subsequent rounds, as shown in <xref ref-type="fig" rid="fig-4">Fig. 4e</xref>. In contrast, FedProx exhibited a steady improvement in validation accuracy, aligned with the average training accuracy, ultimately reaching 98.00% validation accuracy, as seen in <xref ref-type="fig" rid="fig-4">Fig. 4f</xref>. A common trend observed in <xref ref-type="fig" rid="fig-4">Fig. 4</xref> is that validation accuracy consistently outperformed training accuracy. This can be attributed to regularization techniques, such as dropout and L2 regularization, applied during training. Furthermore, the validation accuracy reflects the performance of the aggregated model, which is inherently more robust than individual client models, contributing to its superior validation performance.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Overview of local training and centralized validation performance using FedProx and FedAvg aggregation strategies on the HF dataset. (<bold>a</bold>) Local Train Accuracy with FedAvg; (<bold>b</bold>) Local Train Accuracy with FedProx; (<bold>c</bold>) Local Train Loss with FedAvg; (<bold>d</bold>) Local Train Loss with FedProx; (<bold>e</bold>) Avg. Train vs. Global Validation Accuracy with FedAvg; (<bold>f</bold>) Avg. Train vs. Global Validation Accuracy with FedProx; (<bold>g</bold>) Avg. Train vs. Global Validation Loss with FedAvg; (<bold>h</bold>) Avg. Train vs. Global Validation Loss with FedProx</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67589-fig-4a.tif"/>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67589-fig-4b.tif"/>
</fig>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Ablation Study</title>
<p>To validate the effectiveness of the model&#x2019;s architecture, we conducted an ablation study focusing on the dual-stream architecture and the attention block, which are the core components of our proposed model. <xref ref-type="table" rid="table-4">Table 4</xref> summarizes the results of various configurations evaluated on the test set. The findings show a significant decline in performance across all metrics when attention blocks are excluded or partially included. Without any attention blocks, the model achieved the lowest accuracy of 94.00% with mono-stream and 93.50% with dual-stream. However, the incremental inclusion of attention blocks led to improved performance in both mono- and dual-stream formations. The proposed architecture, incorporating all three attention blocks in the dual-stream formation for effective feature extraction, achieved a test accuracy of 97.50% and an AUC score of 0.9958, underscoring the crucial role of the attention mechanism and the effectiveness of multi-stream feature extraction in enhancing the model&#x2019;s performance.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Summary of the test set performance during ablation study of the proposed model (&#x2717; <inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:mo stretchy="false">&#x2192;</mml:mo></mml:math></inline-formula> none, Acc. &#x003D; Accuracy, P &#x003D; Pecision, R &#x003D; Recall)</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="center">Stream</th>
<th align="center">Attention block</th>
<th align="center">Params <bold>(M)</bold></th>
<th align="center">Acc. <bold>(%)</bold></th>
<th align="center">P <bold>(%)</bold></th>
<th align="center">R <bold>(%)</bold></th>
<th align="center">F1-score</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td>Mono</td>
<td>&#x2717;</td>
<td>0.701</td>
<td>94.00</td>
<td>94.08</td>
<td>94.00</td>
<td>0.9400</td>
<td>0.9823</td>
</tr>
<tr>
<td></td>
<td>Only first block</td>
<td>0.713</td>
<td>95.00</td>
<td>95.00</td>
<td>95.07</td>
<td>0.9499</td>
<td>0.9864</td>
</tr>
<tr>
<td></td>
<td>First two blocks</td>
<td>0.763</td>
<td>94.50</td>
<td>94.71</td>
<td>94.67</td>
<td>0.9450</td>
<td>0.9921</td>
</tr>
<tr>
<td></td>
<td>All three blocks</td>
<td>0.961</td>
<td>96.00</td>
<td>95.99</td>
<td>95.99</td>
<td>0.9599</td>
<td>0.9923</td>
</tr>
<tr>
<td>Dual</td>
<td>&#x2717;</td>
<td>0.843</td>
<td>93.50</td>
<td>93.50</td>
<td>93.50</td>
<td>0.9350</td>
<td>0.9763</td>
</tr>
<tr>
<td></td>
<td>Only first block</td>
<td>0.856</td>
<td>94.50</td>
<td>94.50</td>
<td>94.50</td>
<td>0.9450</td>
<td>0.9878</td>
</tr>
<tr>
<td></td>
<td>First two blocks</td>
<td>0.906</td>
<td>96.00</td>
<td>96.08</td>
<td>96.00</td>
<td>0.9600</td>
<td>0.9936</td>
</tr>
<tr>
<td></td>
<td>All three blocks</td>
<td><bold>1.104</bold></td>
<td><bold>97.50</bold></td>
<td><bold>97.50</bold></td>
<td><bold>97.50</bold></td>
<td><bold>0.9750</bold></td>
<td><bold>0.9958</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Performance Comparision</title>
<p>To benchmark the performance of the proposed model against established video classification approaches, we conducted experiments using transfer learning with two widely recognized transformer-based video classifiers: ViViT [<xref ref-type="bibr" rid="ref-41">41</xref>] and TimeSformer [<xref ref-type="bibr" rid="ref-42">42</xref>]. For a fair comparison, both baselines were trained under the same experimental setup (input resolution, frame sampling, preprocessing, splits, and federated configuration) and optimized with the FedProx aggregation strategy, since our proposed model achieved its best performance with FedProx. We employed an additive fine-tuning approach with their pre-trained base architectures. Specifically, we froze the pre-trained layers of each model and appended two trainable dense layers at the end, incorporating a GELU non-linearity between them. This configuration added over one million trainable parameters to each model, enabling task-specific learning while leveraging the representational power of the pre-trained transformers. <xref ref-type="table" rid="table-5">Table 5</xref> presents the performance metrics obtained from the experiments. Among the baseline models, TimeSformer achieved the highest test accuracy of 96.00% for the HF dataset and 97.05% for the WVD dataset, highlighting its strength in video classification tasks. However, these results slightly lag behind the performance of our proposed model, emphasizing the effectiveness of our approach. Similarly, ViViT recorded the lowest F1-score of 0.945 for HF and 0.9101 for WVD, compared to 0.975 and 0.9699 achieved by our model. Both ViViT and TimeSformer showed significantly poorer results for the Firearm dataset, while our proposed model attained over 96% accuracy. <xref ref-type="fig" rid="fig-5">Fig. 5</xref> provides a visual comparison of the F1-scores produced by these models, demonstrating the superior capability of the proposed model in recognizing violent videos.</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Overview of the test set performance of the fine-tuned baseline models</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Model</th>
<th>Params (M)</th>
<th>Dataset</th>
<th>Accuracy (%)</th>
<th>Precision (%)</th>
<th>Recall (%)</th>
<th>F1-score</th>
<th>AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td>ViViT</td>
<td>554.2</td>
<td>HF</td>
<td>94.50</td>
<td>94.50</td>
<td>94.50</td>
<td>0.9450</td>
<td>0.9905</td>
</tr>
<tr>
<td></td>
<td></td>
<td>WVD</td>
<td>94.12</td>
<td>96.43</td>
<td>87.50</td>
<td>0.9101</td>
<td>0.9856</td>
</tr>
<tr>
<td></td>
<td></td>
<td>Firearm</td>
<td>80.00</td>
<td>85.48</td>
<td>77.59</td>
<td>0.7919</td>
<td>0.9432</td>
</tr>
<tr>
<td>TimeSformer</td>
<td>777.2</td>
<td>HF</td>
<td>96.00</td>
<td>96.02</td>
<td>96.00</td>
<td>0.9600</td>
<td>0.9915</td>
</tr>
<tr>
<td></td>
<td></td>
<td>WVD</td>
<td>97.05</td>
<td>98.14</td>
<td>93.75</td>
<td>0.9572</td>
<td><bold>0.9903</bold></td>
</tr>
<tr>
<td></td>
<td></td>
<td>Firearm</td>
<td>78.75</td>
<td>78.79</td>
<td>78.75</td>
<td>0.7872</td>
<td>0.8962</td>
</tr>
<tr>
<td><bold>Proposed</bold></td>
<td>1.104</td>
<td>HF</td>
<td><bold>97.50</bold></td>
<td><bold>97.50</bold></td>
<td><bold>97.50</bold></td>
<td><bold>0.9750</bold></td>
<td><bold>0.9958</bold></td>
</tr>
<tr>
<td></td>
<td></td>
<td>WVD</td>
<td><bold>97.06</bold></td>
<td><bold>97.17</bold></td>
<td><bold>97.06</bold></td>
<td><bold>0.9699</bold></td>
<td>0.9616</td>
</tr>
<tr>
<td></td>
<td></td>
<td>Firearm</td>
<td><bold>96.25</bold></td>
<td><bold>96.30</bold></td>
<td><bold>96.25</bold></td>
<td><bold>0.9625</bold></td>
<td><bold>0.9952</bold></td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>F1-score visual comparison among baseline models and the proposed one</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67589-fig-5.tif"/>
</fig>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Discussion</title>
<p>This study introduces a DL-based method for violent activity recognition that prioritizes user data privacy by employing FL and computational efficiency by incorporating a lightweight design tailored for edge deployment. With only approximately 1.104 million parameters and a model size of 4.42 MB, our model achieves competitive performance, with test set accuracies ranging from 96.25% to 97.50%. This strong performance is a result of our efficient architectural design, which integrates lightweight operations such as depthwise separable 3D convolutions and linear attention mechanisms, and notably the use of the FedProx aggregation method. However,</p>
<p>While our study demonstrates promising results, this study focuses on simulating the feasibility of lightweight models within FL frameworks for violent activity recognition and the key next step is to validate the proposed model on representative edge hardware and quantify practical deployment trade-offs. Future studies with large-scale and more diverse datasets will enhance the statistical reliability and generalizability of our findings. Moreover, although FedProx can effectively handle challenges posed by heterogeneous data distribution via its proximal penalty term, extended experiments with more clients and varying data distributions will better reflect the robustness of our model and FL strategy against the heterogeneity of real-world surveillance systems. In addition, the exploration of the impact of tailored augmentations specific to violent activity characteristics and the role of varying experimental setups (e.g., varying batch sizes) is left for future research. On the other hand, models can sometimes learn spurious or irrelevant patterns while still producing correct predictions; integrating explainable AI (XAI) techniques (e.g., attention-map visualizations, and gradient-based methods) is an important future direction to provide impactful qualitative analyses and to clarify model decision-making, thereby increasing the trustworthiness of HAR applications.</p>
<p>By pursuing these directions, we anticipate establishing a versatile, privacy-preserving and trustworthy activity recognition framework that can adapt to a wide range of applications&#x2014;from public safety monitoring to assistive care, while maintaining the stringent data-protection guarantees required in sensitive environments.</p>
</sec>
<sec id="s6">
<label>6</label>
<title>Conclusion</title>
<p>We propose an effective and efficient DL model for automated violent activity recognition from videos, designed with a focus on real-world, resource-constrained applications. Our approach emphasizes decentralized training, eliminating reliance on cloud servers, and ensuring enhanced user data privacy by keeping data confined to the native device. The proposed lightweight architecture achieved competitive test set accuracies across multiple experimental datasets, outperforming larger pre-trained models that have more than 100 times the number of parameters. These findings not only highlight the model&#x2019;s effectiveness but also demonstrate its generalization potential across diverse scenarios. Additionally, we found that the FedProx aggregation strategy improves the performance of the employed model over the FedAvg strategy. This study puts a significant step toward developing privacy-enhancing and efficient solutions for violent activity recognition, deployment on edge devices, and beyond, facilitating the broader adoption of secure and effective AI systems in sensitive applications.</p>
</sec>
</body>
<back>
<ack>
<p>The authors extend their appreciation to the Research Chair of Online Dialogue and Cultural Communication, King Saud University, Saudi Arabia, for funding this research.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>This work was supported by the Research Chair of Online Dialogue and Cultural Communication, King Saud University, Saudi Arabia.</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>The authors confirm contribution to the paper as follows: Conceptualization, Moshiur Rahman Tonmoy; methodology, Moshiur Rahman Tonmoy and Md. Mithun Hossain; software, Moshiur Rahman Tonmoy and Md. Mithun Hossain; validation, Mejdl Safran, Sultan Alfarhood and M. F. Mridha; formal analysis, Md. Mithun Hossain; investigation, Moshiur Rahman Tonmoy and Md. Mithun Hossain; resources, Mejdl Safran and M. F. Mridha; data curation, Md. Mithun Hossain; writing&#x2014;original draft preparation, Moshiur Rahman Tonmoy and Md. Mithun Hossain; writing&#x2014;review and editing, Mejdl Safran, Dunren Che and M. F. Mridha; visualization, Dunren Che and Sultan Alfarhood; supervision, M. F. Mridha; project administration, M. F. Mridha; funding acquisition, Mejdl Safran and Sultan Alfarhood. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>The data that support the findings of this study are openly available in the respective repository as referenced in this study.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chaudhary</surname> <given-names>D</given-names></string-name>, <string-name><surname>Kumar</surname> <given-names>S</given-names></string-name>, <string-name><surname>Dhaka</surname> <given-names>VS</given-names></string-name></person-group>. <article-title>Video based human crowd analysis using machine learning: a survey</article-title>. <source>Comput Methods Biomech Biomedical Eng Imag Visual</source>. <year>2022</year>;<volume>10</volume>(<issue>2</issue>):<fpage>113</fpage>&#x2013;<lpage>31</lpage>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tripathi</surname> <given-names>G</given-names></string-name>, <string-name><surname>Singh</surname> <given-names>K</given-names></string-name>, <string-name><surname>Vishwakarma</surname> <given-names>DK</given-names></string-name></person-group>. <article-title>Convolutional neural networks for crowd behaviour analysis: a survey</article-title>. <source>Vis Comput</source>. <year>2019</year>;<volume>35</volume>(<issue>5</issue>):<fpage>753</fpage>&#x2013;<lpage>76</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s00371-018-1499-5</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><collab>MarketsandMarkets</collab></person-group>. <article-title>Video surveillance industry worth $88.71 billion by 2030; 2025 [Internet]. [cited 2025 Jul 22]</article-title>. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.marketsandmarkets.com/PressReleases/global-video-surveillance-market.asp">https://www.marketsandmarkets.com/PressReleases/global-video-surveillance-market.asp</ext-link>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Badidi</surname> <given-names>E</given-names></string-name>, <string-name><surname>Moumane</surname> <given-names>K</given-names></string-name>, <string-name><surname>El Ghazi</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Opportunities, applications, and challenges of edge-ai enabled video analytics in smart cities: a systematic review</article-title>. <source>IEEE Access</source>. <year>2023</year>;<volume>11</volume>:<fpage>80543</fpage>&#x2013;<lpage>72</lpage>. doi:<pub-id pub-id-type="doi">10.1109/access.2023.3300658</pub-id>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Teng</surname> <given-names>L</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Privacy-preserving video anomaly detection: a survey</article-title>. <comment>arXiv:2411.14565. 2025</comment>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>ElBaih</surname> <given-names>M</given-names></string-name></person-group>. <article-title>The role of privacy regulations in AI development (A discussion of the ways in which privacy regulations can shape the development of AI)</article-title>. <comment>2023</comment>. [cited 2025 Jul 22]. Available from: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.2139/ssrn.4589207">http://dx.doi.org/10.2139/ssrn.4589207</ext-link>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gupta</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Safeguarding digital privacy with ai-driven solutions</article-title>. <source>ESP Int J Adv Comput Technol (ESP-IJACT)</source>. <year>2024</year>;<volume>2</volume>(<issue>1</issue>):<fpage>126</fpage>&#x2013;<lpage>42</lpage>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Do</surname> <given-names>TTT</given-names></string-name>, <string-name><surname>Huynh</surname> <given-names>QT</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>K</given-names></string-name>, <string-name><surname>Nguyen</surname> <given-names>VQ</given-names></string-name></person-group>. <article-title>A survey on video big data analytics: architecture, technologies, and open research challenges</article-title>. <source>Appl Sci</source>. <year>2025</year>;<volume>15</volume>(<issue>14</issue>):<fpage>8089</fpage>. doi:<pub-id pub-id-type="doi">10.3390/app15148089</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>T</given-names></string-name>, <string-name><surname>Tong</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Federated machine learning: concept and applications</article-title>. <source>ACM Trans Intell Syst Technol</source>. <year>2019</year>;<volume>10</volume>(<issue>2</issue>):<fpage>12</fpage>. doi:<pub-id pub-id-type="doi">10.1145/3298981</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Shenaj</surname> <given-names>D</given-names></string-name>, <string-name><surname>Rizzoli</surname> <given-names>G</given-names></string-name>, <string-name><surname>Zanuttigh</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Federated learning in computer vision</article-title>. <source>IEEE Access</source>. <year>2023</year>;<volume>11</volume>:<fpage>94863</fpage>&#x2013;<lpage>84</lpage>. doi:<pub-id pub-id-type="doi">10.1109/access.2023.3310400</pub-id>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Alsmirat</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Obaidat</surname> <given-names>I</given-names></string-name>, <string-name><surname>Jararweh</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Al-Saleh</surname> <given-names>M</given-names></string-name></person-group>. <article-title>A security framework for cloud-based video surveillance system</article-title>. <source>Multimedia Tools Appl</source>. <year>2017</year>;<volume>76</volume>(<issue>21</issue>):<fpage>22787</fpage>&#x2013;<lpage>802</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s11042-017-4488-1</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Shifa</surname> <given-names>A</given-names></string-name>, <string-name><surname>Asghar</surname> <given-names>MN</given-names></string-name>, <string-name><surname>Noor</surname> <given-names>S</given-names></string-name>, <string-name><surname>Gohar</surname> <given-names>N</given-names></string-name>, <string-name><surname>Fleury</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Lightweight cipher for H.264 videos in the Internet of multimedia things with encryption space ratio diagnostics</article-title>. <source>Sensors</source>. <year>2019</year>;<volume>19</volume>(<issue>5</issue>):<fpage>1228</fpage>. doi:<pub-id pub-id-type="doi">10.3390/s19051228</pub-id>; <pub-id pub-id-type="pmid">30862057</pub-id></mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bagdasaryan</surname> <given-names>E</given-names></string-name>, <string-name><surname>Poursaeed</surname> <given-names>O</given-names></string-name>, <string-name><surname>Shmatikov</surname> <given-names>V</given-names></string-name></person-group>. <article-title>Differential privacy has disparate impact on model accuracy</article-title>. <source>Adv Neural Inf Process Syst</source>. <year>2019</year>;<volume>32</volume>. [cited 2025 Jul 12]. Available from: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2019/hash/fc0de4e0396fff257ea362983c2dda5a-Abstract.html">https://proceedings.neurips.cc/paper/2019/hash/fc0de4e0396fff257ea362983c2dda5a-Abstract.html</ext-link>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tonmoy</surname> <given-names>MR</given-names></string-name>, <string-name><surname>Rakib</surname> <given-names>AF</given-names></string-name>, <string-name><surname>Rahman</surname> <given-names>R</given-names></string-name>, <string-name><surname>Adnan</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Mridha</surname> <given-names>MF</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A lightweight visual font style recognition with quantized convolutional autoencoder</article-title>. <source>IEEE Open J Comput Soc</source>. <year>2024</year>;<volume>5</volume>(<issue>2</issue>):<fpage>120</fpage>&#x2013;<lpage>30</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ojcs.2024.3378709</pub-id>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ullah</surname> <given-names>FUM</given-names></string-name>, <string-name><surname>Obaidat</surname> <given-names>MS</given-names></string-name>, <string-name><surname>Ullah</surname> <given-names>A</given-names></string-name>, <string-name><surname>Muhammad</surname> <given-names>K</given-names></string-name>, <string-name><surname>Hijji</surname> <given-names>M</given-names></string-name>, <string-name><surname>Baik</surname> <given-names>SW</given-names></string-name></person-group>. <article-title>A comprehensive review on vision-based violence detection in surveillance videos</article-title>. <source>ACM Comput Surv</source>. <year>2023</year>;<volume>55</volume>(<issue>10</issue>):<fpage>1</fpage>&#x2013;<lpage>44</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3561971</pub-id>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Fu</surname> <given-names>W</given-names></string-name></person-group>. <article-title>Human-centered attention-aware networks for action recognition</article-title>. <source>Int J Intell Syst</source>. <year>2022</year>;<volume>37</volume>(<issue>12</issue>):<fpage>10968</fpage>&#x2013;<lpage>87</lpage>. doi:<pub-id pub-id-type="doi">10.1002/int.23029</pub-id>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Aggarwal</surname> <given-names>S</given-names></string-name>, <string-name><surname>Ranjan</surname> <given-names>R</given-names></string-name>, <string-name><surname>Sinha</surname> <given-names>M</given-names></string-name>, <string-name><surname>Pal</surname> <given-names>V</given-names></string-name>, <string-name><surname>Kushwaha</surname> <given-names>R</given-names></string-name></person-group>. <article-title>CNN and BiLSTM based framework for real life violence detection from CCTV videos</article-title>. In: <conf-name>2024 IEEE Region 10 Symposium (TENSYMP); 2024 Sep 27&#x2013;29; New Delhi, India</conf-name>. p. <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yadav</surname> <given-names>V</given-names></string-name>, <string-name><surname>Kumar</surname> <given-names>S</given-names></string-name>, <string-name><surname>Goyal</surname> <given-names>A</given-names></string-name>, <string-name><surname>Bhatla</surname> <given-names>S</given-names></string-name>, <string-name><surname>Sikka</surname> <given-names>G</given-names></string-name>, <string-name><surname>Kaur</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Integrated violence and weapon detection using deep learning</article-title>. In: <conf-name>2024 First International Conference on Pioneering Developments in Computer Science &#x0026; Digital Technologies (IC2SDT); 2024 Aug 2&#x2013;4; Delhi, India</conf-name>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Abbass</surname> <given-names>MAB</given-names></string-name>, <string-name><surname>Kang</surname> <given-names>HS</given-names></string-name></person-group>. <article-title>Violence detection enhancement by involving convolutional block attention modules into various deep learning architectures: comprehensive case study for UBI-fights dataset</article-title>. <source>IEEE Access</source>. <year>2023</year>;<volume>11</volume>:<fpage>37096</fpage>&#x2013;<lpage>107</lpage>. doi:<pub-id pub-id-type="doi">10.1109/access.2023.3267409</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Kumar</surname> <given-names>A</given-names></string-name>, <string-name><surname>Shetty</surname> <given-names>A</given-names></string-name>, <string-name><surname>Sagar</surname> <given-names>A</given-names></string-name>, <string-name><surname>Charushree</surname> <given-names>A</given-names></string-name>, <string-name><surname>Kanwal</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Indoor violence detection using lightweight transformer model</article-title>. In: <conf-name>2023 4th International Conference for Emerging Technology (INCET); 2023 May 26&#x2013;28; Belgaum, India</conf-name>. p. <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mohammadi</surname> <given-names>H</given-names></string-name>, <string-name><surname>Nazerfard</surname> <given-names>E</given-names></string-name></person-group>. <article-title>Video violence recognition and localization using a semi-supervised hard attention model</article-title>. <source>Expert Syst Appl</source>. <year>2023</year>;<volume>212</volume>:<fpage>118791</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.eswa.2022.118791</pub-id>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Vijeikis</surname> <given-names>R</given-names></string-name>, <string-name><surname>Raudonis</surname> <given-names>V</given-names></string-name>, <string-name><surname>Dervinis</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Efficient violence detection in surveillance</article-title>. <source>Sensors</source>. <year>2022</year>;<volume>22</volume>(<issue>6</issue>):<fpage>2216</fpage>. doi:<pub-id pub-id-type="doi">10.3390/s22062216</pub-id>; <pub-id pub-id-type="pmid">35336387</pub-id></mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Frimpong</surname> <given-names>E</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>T</given-names></string-name>, <string-name><surname>Michalas</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Secrets in motion: privacy-preserving video classification with built-in access control</article-title>. In: <conf-name>2024 9th International Conference on Smart and Sustainable Technologies (SpliTech); 2024 Jun 25&#x2013;28; Bol and Split, Croatia</conf-name>. p. <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Feng</surname> <given-names>D</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>S</given-names></string-name>, <string-name><surname>Tung</surname> <given-names>L</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>F</given-names></string-name></person-group>. <article-title>X-stream: a flexible, adaptive video transformer for privacy-preserving video stream analytics</article-title>. In: <conf-name>IEEE INFOCOM 2024-IEEE Conference on Computer Communications; 2024 May 20&#x2013;23; Vancouver, BC, Canada</conf-name>. p. <fpage>1</fpage>&#x2013;<lpage>10</lpage>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gaikwad</surname> <given-names>B</given-names></string-name>, <string-name><surname>Karmakar</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Real-time distributed video analytics for privacy-aware person search</article-title>. <source>Comput Vis Image Underst</source>. <year>2023</year>;<volume>234</volume>(<issue>3</issue>):<fpage>103749</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.cviu.2023.103749</pub-id>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Mehta</surname> <given-names>D</given-names></string-name>, <string-name><surname>Sivathamboo</surname> <given-names>S</given-names></string-name>, <string-name><surname>Simpson</surname> <given-names>H</given-names></string-name>, <string-name><surname>Kwan</surname> <given-names>P</given-names></string-name>, <string-name><surname>O&#x2019;Brien</surname> <given-names>T</given-names></string-name>, <string-name><surname>Ge</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>Privacy-preserving early detection of epileptic seizures in videos</article-title>. In: <conf-name>International Conference on Medical Image Computing and Computer-Assisted Intervention</conf-name>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer Nature</publisher-name>; <year>2023</year>. p. <fpage>210</fpage>&#x2013;<lpage>9</lpage>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Singh</surname> <given-names>S</given-names></string-name>, <string-name><surname>Dewangan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Krishna</surname> <given-names>GS</given-names></string-name>, <string-name><surname>Tyagi</surname> <given-names>V</given-names></string-name>, <string-name><surname>Reddy</surname> <given-names>S</given-names></string-name>, <string-name><surname>Medi</surname> <given-names>PR</given-names></string-name></person-group>. <article-title>Video vision transformers for violence detection</article-title>. <comment>arXiv:2209.03561. 2022</comment>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Pajon</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Serre</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wissocq</surname> <given-names>H</given-names></string-name>, <string-name><surname>Rabaud</surname> <given-names>L</given-names></string-name>, <string-name><surname>Haidar</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yaacoub</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Balancing accuracy and training time in federated learning for violence detection in surveillance videos: a study of neural network architectures</article-title>. <source>J Comput Sci Technol</source>. <year>2024</year>;<volume>39</volume>(<issue>5</issue>):<fpage>1029</fpage>&#x2013;<lpage>39</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s11390-024-3702-7</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Victor</surname> <given-names>EDS</given-names></string-name>, <string-name><surname>Lacerda</surname> <given-names>TB</given-names></string-name>, <string-name><surname>Miranda</surname> <given-names>PB</given-names></string-name>, <string-name><surname>Nascimento</surname> <given-names>AC</given-names></string-name>, <string-name><surname>Furtado</surname> <given-names>APC</given-names></string-name></person-group>. <article-title>Federated learning for physical violence detection in videos</article-title>. In: <conf-name>2022 International Joint Conference on Neural Networks (IJCNN); 2022 Jul 18&#x2013;23; Padua, Italy</conf-name>. p. <fpage>1</fpage>&#x2013;<lpage>8</lpage>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>McMahan</surname> <given-names>B</given-names></string-name>, <string-name><surname>Moore</surname> <given-names>E</given-names></string-name>, <string-name><surname>Ramage</surname> <given-names>D</given-names></string-name>, <string-name><surname>Hampson</surname> <given-names>S</given-names></string-name>, <string-name><surname>BAy</surname> <given-names>Arcas</given-names></string-name></person-group>. <chapter-title>Communication-efficient learning of deep networks from decentralized data</chapter-title>. In: <person-group person-group-type="editor"><string-name><surname>Singh</surname> <given-names>A</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>J</given-names></string-name></person-group>, editors. <source>Proceedings of the 20th international conference on artificial intelligence and statistics</source>. Vol. <volume>54</volume>. <publisher-loc>Westminster, UK</publisher-loc>: <publisher-name>PMLR</publisher-name>; <year>2017</year>. p. <fpage>1273</fpage>&#x2013;<lpage>82</lpage>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Guan</surname> <given-names>H</given-names></string-name>, <string-name><surname>Yap</surname> <given-names>PT</given-names></string-name>, <string-name><surname>Bozoki</surname> <given-names>A</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Federated learning for medical image analysis: a survey</article-title>. <source>Pattern Recognit</source>. <year>2024</year>;<volume>151</volume>(<issue>3</issue>):<fpage>110424</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.patcog.2024.110424</pub-id>; <pub-id pub-id-type="pmid">38559674</pub-id></mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Mistry</surname> <given-names>D</given-names></string-name>, <string-name><surname>Tonmoy</surname> <given-names>MR</given-names></string-name>, <string-name><surname>Anower</surname> <given-names>MS</given-names></string-name>, <string-name><surname>Hasan</surname> <given-names>ASMT</given-names></string-name></person-group>. <chapter-title>Federated transfer learning for vision-based fall detection</chapter-title>. In: <person-group person-group-type="editor"><string-name><surname>Arefin</surname> <given-names>MS</given-names></string-name>, <string-name><surname>Kaiser</surname> <given-names>MS</given-names></string-name>, <string-name><surname>Bhuiyan</surname> <given-names>T</given-names></string-name>, <string-name><surname>Dey</surname> <given-names>N</given-names></string-name>, <string-name><surname>Mahmud</surname> <given-names>M</given-names></string-name></person-group>, editors. <source>Proceedings of the 2nd International Conference on
Big Data, IoT and Machine Learning</source>. <publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer Nature Singapore</publisher-name>; <year>2024</year>. p. <fpage>961</fpage>&#x2013;<lpage>75</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-981-99-8937-9_64</pub-id>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>T</given-names></string-name>, <string-name><surname>Sahu</surname> <given-names>AK</given-names></string-name>, <string-name><surname>Zaheer</surname> <given-names>M</given-names></string-name>, <string-name><surname>Sanjabi</surname> <given-names>M</given-names></string-name>, <string-name><surname>Talwalkar</surname> <given-names>A</given-names></string-name>, <string-name><surname>Smith</surname> <given-names>V</given-names></string-name></person-group>. <article-title>Federated optimization in heterogeneous networks</article-title>. In: <conf-name>Proceedings of Machine Learning and Systems; 2020 Mar 2&#x2013;4; Austin, TX, USA</conf-name>. p. <fpage>429</fpage>&#x2013;<lpage>50</lpage>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tonmoy</surname> <given-names>MR</given-names></string-name>, <string-name><surname>Shams</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Adnan</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Mridha</surname> <given-names>MF</given-names></string-name>, <string-name><surname>Safran</surname> <given-names>M</given-names></string-name>, <string-name><surname>Alfarhood</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>X-Brain: explainable recognition of brain tumors using robust deep attention CNN</article-title>. <source>Biomed Signal Process Control</source>. <year>2025</year>;<volume>100</volume>(<issue>18</issue>):<fpage>106988</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.bspc.2024.106988</pub-id>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Hendrycks</surname> <given-names>D</given-names></string-name>, <string-name><surname>Gimpel</surname> <given-names>K</given-names></string-name></person-group>. <article-title>Gaussian error linear units (GELUs)</article-title>. <comment>arXiv:1606.08415. 2023</comment>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mehta</surname> <given-names>S</given-names></string-name>, <string-name><surname>Rastegari</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Separable self-attention for mobile vision transformers</article-title>. <source>Trans Mach Learn Res</source>. <year>2023</year>. [cited 2025 Jul 12]. <ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=tBl4yBEjKi">https://openreview.net/forum?id=tBl4yBEjKi</ext-link>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Bermejo Nievas</surname> <given-names>E</given-names></string-name>, <string-name><surname>Deniz Suarez</surname> <given-names>O</given-names></string-name>, <string-name><surname>Bueno Garc&#x00ED;a</surname> <given-names>G</given-names></string-name>, <string-name><surname>Sukthankar</surname> <given-names>R</given-names></string-name></person-group>. <chapter-title>Violence detection in video using computer vision techniques</chapter-title>. In: <person-group person-group-type="editor"><string-name><surname>Real</surname> <given-names>P</given-names></string-name>, <string-name><surname>Diaz-Pernil</surname> <given-names>D</given-names></string-name>, <string-name><surname>Molina-Abril</surname> <given-names>H</given-names></string-name>, <string-name><surname>Berciano</surname> <given-names>A</given-names></string-name>, <string-name><surname>Kropatsch</surname> <given-names>W</given-names></string-name></person-group>, editors. <source>Computer analysis of images and patterns</source>. <publisher-loc>Berlin/Heidelberg, Germany</publisher-loc>: <publisher-name>Springer</publisher-name>; <year>2011</year>. p. <fpage>332</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.1007/978-3-642-23678-5_39</pub-id>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Nadeem</surname> <given-names>MS</given-names></string-name>, <string-name><surname>Kurugollu</surname> <given-names>F</given-names></string-name>, <string-name><surname>Atlam</surname> <given-names>HF</given-names></string-name>, <string-name><surname>Franqueira</surname> <given-names>VNL</given-names></string-name></person-group>. <article-title>Weapon violence dataset 2.0: a synthetic dataset for violence detection</article-title>. <source>Data Brief</source>. <year>2024</year>;<volume>54</volume>(<issue>8</issue>):<fpage>110448</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.dib.2024.110448</pub-id>; <pub-id pub-id-type="pmid">38725552</pub-id></mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ruiz-Santaquiteria</surname> <given-names>J</given-names></string-name>, <string-name><surname>Mu&#x00F1;oz</surname> <given-names>JD</given-names></string-name>, <string-name><surname>Maigler</surname> <given-names>FJ</given-names></string-name>, <string-name><surname>Deniz</surname> <given-names>O</given-names></string-name>, <string-name><surname>Bueno</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Firearm-related action recognition and object detection dataset for video surveillance systems</article-title>. <source>Data Brief</source>. <year>2024</year>;<volume>52</volume>(<issue>24</issue>):<fpage>110030</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.dib.2024.110030</pub-id>; <pub-id pub-id-type="pmid">38299104</pub-id></mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Buslaev</surname> <given-names>A</given-names></string-name>, <string-name><surname>Iglovikov</surname> <given-names>VI</given-names></string-name>, <string-name><surname>Khvedchenya</surname> <given-names>E</given-names></string-name>, <string-name><surname>Parinov</surname> <given-names>A</given-names></string-name>, <string-name><surname>Druzhinin</surname> <given-names>M</given-names></string-name>, <string-name><surname>Kalinin</surname> <given-names>AA</given-names></string-name></person-group>. <article-title>Albumentations: fast and flexible image augmentations</article-title>. <source>Information</source>. <year>2020</year>;<volume>11</volume>(<issue>2</issue>):<fpage>125</fpage>. doi:<pub-id pub-id-type="doi">10.3390/info11020125</pub-id>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Arnab</surname> <given-names>A</given-names></string-name>, <string-name><surname>Dehghani</surname> <given-names>M</given-names></string-name>, <string-name><surname>Heigold</surname> <given-names>G</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>C</given-names></string-name>, <string-name><surname>Lu&#x010D;i&#x0107;</surname> <given-names>M</given-names></string-name>, <string-name><surname>Schmid</surname> <given-names>C</given-names></string-name></person-group>. <article-title>ViViT: a video vision transformer</article-title>. In: <conf-name>Proceedings of the 2021 IEEE/CVF International Conference on Computer Vision (ICCV); 2021 Oct 10&#x2013;17; Montreal, QC, Canada</conf-name>. p. <fpage>6836</fpage>&#x2013;<lpage>46</lpage>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Bertasius</surname> <given-names>G</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Torresani</surname> <given-names>L</given-names></string-name></person-group>. <chapter-title>Is space-time attention all you need for video understanding?</chapter-title>. In: <person-group person-group-type="editor"><string-name><surname>Meila</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>T</given-names></string-name></person-group>, editors. <source>Proceedings of the 38th international conference on machine learning</source>. Vol. <volume>139</volume>. <publisher-loc>Westminster, UK</publisher-loc>: <publisher-name>PMLR</publisher-name>; <year>2021</year>. p. <fpage>813</fpage>&#x2013;<lpage>24</lpage>.</mixed-citation></ref>
</ref-list>
</back></article>







