<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">63563</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2025.063563</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>IoT-Based Real-Time Medical-Related Human Activity Recognition Using Skeletons and Multi-Stage Deep Learning for Healthcare</article-title>
<alt-title alt-title-type="left-running-head">IoT-Based Real-Time Medical-Related Human Activity Recognition Using Skeletons and Multi-Stage Deep Learning for Healthcare</alt-title>
<alt-title alt-title-type="right-running-head">IoT-Based Real-Time Medical-Related Human Activity Recognition Using Skeletons and Multi-Stage Deep Learning for Healthcare</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Paul</surname><given-names>Subrata Kumer</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Miah</surname><given-names>Abu Saleh Musa</given-names></name><xref ref-type="aff" rid="aff-3">3</xref><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Paul</surname><given-names>Rakhi Rani</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Hamid</surname><given-names>Md. Ekramul</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-5" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Shin</surname><given-names>Jungpil</given-names></name><xref ref-type="aff" rid="aff-4">4</xref><email>jpshin@u-aizu.ac.jp</email></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Rahim</surname><given-names>Md Abdur</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<aff id="aff-1"><label>1</label><institution>Department of Computer Science and Engineering (CSE), Bangladesh Army University of Engineering &#x0026; Technology (BAUET)</institution>, <addr-line>Qadirabad Cantonment, Dayarampur, Natore, 6431</addr-line>, <country>Bangladesh</country></aff>
<aff id="aff-2"><label>2</label><institution>Department of Computer Science and Engineering (CSE), Rajshahi University</institution>, <addr-line>Rajshahi, 6205</addr-line>, <country>Bangladesh</country></aff>
<aff id="aff-3"><label>3</label><institution>Department of CSE, Bangladesh Army University of Science and Technology (BAUST)</institution>, <addr-line>Nilphamari, Saidpur, 5311</addr-line>, <country>Bangladesh</country></aff>
<aff id="aff-4"><label>4</label><institution>School of Computer Science and Engineering, The University of Aizu</institution>, <addr-line>Aizuwakmatsu, 965</addr-line>-<addr-line>8580</addr-line>, <country>Japan</country></aff>
<aff id="aff-5"><label>5</label><institution>Department of Computer Science and Engineering, Pabna University of Science and Technology</institution>, <addr-line>Rajapur, 6600</addr-line>, <country>Bangladesh</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Jungpil Shin. Email: <email>jpshin@u-aizu.ac.jp</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2025</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>03</day><month>07</month><year>2025</year>
</pub-date>
<volume>84</volume>
<issue>2</issue>
<fpage>2513</fpage>
<lpage>2530</lpage>
<history>
<date date-type="received">
<day>17</day>
<month>1</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>4</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 The Authors.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_63563.pdf"></self-uri>
<abstract>
<p>The Internet of Things (IoT) and mobile technology have significantly transformed healthcare by enabling real-time monitoring and diagnosis of patients. Recognizing Medical-Related Human Activities (MRHA) is pivotal for healthcare systems, particularly for identifying actions critical to patient well-being. However, challenges such as high computational demands, low accuracy, and limited adaptability persist in Human Motion Recognition (HMR). While some studies have integrated HMR with IoT for real-time healthcare applications, limited research has focused on recognizing MRHA as essential for effective patient monitoring. This study proposes a novel HMR method tailored for MRHA detection, leveraging multi-stage deep learning techniques integrated with IoT. The approach employs EfficientNet to extract optimized spatial features from skeleton frame sequences using seven Mobile Inverted Bottleneck Convolutions (MBConv) blocks, followed by Convolutional Long Short Term Memory (ConvLSTM) to capture spatio-temporal patterns. A classification module with global average pooling, a fully connected layer, and a dropout layer generates the final predictions. The model is evaluated on the NTU RGB&#x002B;D 120 and HMDB51 datasets, focusing on MRHA such as sneezing, falling, walking, sitting, etc. It achieves 94.85% accuracy for cross-subject evaluations and 96.45% for cross-view evaluations on NTU RGB&#x002B;D 120, along with 89.22% accuracy on HMDB51. Additionally, the system integrates IoT capabilities using a Raspberry Pi and GSM module, delivering real-time alerts via Twilios SMS service to caregivers and patients. This scalable and efficient solution bridges the gap between HMR and IoT, advancing patient monitoring, improving healthcare outcomes, and reducing costs.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Real-time human motion recognition (HMR)</kwd>
<kwd>ENConvLSTM</kwd>
<kwd>EfficientNet</kwd>
<kwd>ConvLSTM</kwd>
<kwd>skeleton data</kwd>
<kwd>NTU RGB&#x002B;D 120 dataset</kwd>
<kwd>MRHA</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>ICT Division of the Ministry of Posts</funding-source>
<award-id>56.00.0000.052.33.005.21-7</award-id>
</award-group>
<award-group id="awg2">
<funding-source>University of Rajshahi</funding-source>
<award-id>22FS15306</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Human Motion Recognition (HMR) systems aim to automatically identify and monitor individual or group activities, playing a crucial role in healthcare by tracking physical and medical-related activities. Among these, medical-related human activities (MRHA), such as sneezing, coughing, falling, and staggering, are vital for ensuring patient safety and well-being. However, accurate detection of MRHA remains a significant challenge. With the global elderly population expected to reach 2.1 billion by 2050, according to the World Health Organization (WHO) and the United Nations, the demand for effective healthcare solutions is becoming increasingly critical. Many elderly individuals live alone or in care facilities where healthcare professionals are often outnumbered, leading to insufficient monitoring and increased risks. Detecting MRHA, such as falls or signs of distress, is essential to enable real-time interventions and prevent critical incidents. Falls are a particularly significant concern, causing over 646,000 deaths and 37 million severe injuries annually, as reported by the WHO. As the elderly population grows, accounting for approximately 16% of the global population by 2050, the need for automated MRHA detection systems becomes increasingly urgent. These systems enhance patient safety, provide continuous monitoring, and reduce the strain on healthcare professionals [<xref ref-type="bibr" rid="ref-1">1</xref>&#x2013;<xref ref-type="bibr" rid="ref-4">4</xref>]. Medical-Related Human Activity (MRHA) recognition systems are vital for ensuring safety and providing continuous monitoring of in-home care and assisted medical facilities. These systems enable timely interventions in cases of medical emergencies, such as falls, significantly reducing mortality risk by 80% and minimizing extended hospital stays by 26% [<xref ref-type="bibr" rid="ref-5">5</xref>]. Current approaches for MRHA detection primarily use wearable sensors and vision-based methods [<xref ref-type="bibr" rid="ref-3">3</xref>], with sensors capturing acceleration changes associated with falls [<xref ref-type="bibr" rid="ref-6">6</xref>] and vision-based systems analyzing video data [<xref ref-type="bibr" rid="ref-7">7</xref>]. However, developing robust automated MRHA recognition systems is crucial to deliver timely interventions and prevent severe injuries and fatalities. In this study, we selected the NTU RGB&#x002B;D 120 and HMDB51 datasets due to their diverse range of human activities, including several medically relevant motions such as falling, walking, sitting, drinking, and eating, which are crucial for healthcare applications. NTU RGB&#x002B;D 120, in particular, is one of the largest and most widely used action recognition datasets.</p>
<sec id="s1_1">
<label>1.1</label>
<title>Current Fall Detection Systems and Their Challenges</title>
<p>Despite advancements, existing MRHA recognition technologies face significant limitations. Wearable sensor-based systems, though effective in detecting acceleration changes, often struggle with user comfort, false positives during non-critical activities, and low compliance among elderly individuals with cognitive impairments [<xref ref-type="bibr" rid="ref-8">8</xref>]. Vision-based systems provide a non-invasive alternative but raise privacy concerns, as video feeds can compromise individual anonymity and legal safeguards, even when encoding techniques are employed to obscure clarity. Skeleton-based data, derived from pose estimation algorithms [<xref ref-type="bibr" rid="ref-9">9</xref>] or Kinect systems [<xref ref-type="bibr" rid="ref-10">10</xref>], offers a promising solution. Kinect systems are used in human activity recognition (HAR) by capturing 3D skeletal data through depth sensors and tracking key joints to build a model of human posture and movement. This approach preserves privacy by omitting identifiable information while maintaining robustness against challenges such as background noise and lighting variations. Skeleton-based data also have lower dimensionality, ensuring efficient motion representation with reduced computational costs. Recent studies have highlighted the efficacy of skeleton-based MRHA recognition. For instance, Zahan et al. [<xref ref-type="bibr" rid="ref-11">11</xref>] achieved over 94% accuracy on URFD and UPFD datasets using Graph Convolutional Networks (GCNs) combined with Convolutional Neural Networks (CNNs). Similarly, Egawa et al. [<xref ref-type="bibr" rid="ref-2">2</xref>] applied a modified GCN model to the ImVia RU-Fall dataset, reporting a 99.00% accuracy.</p>
</sec>
<sec id="s1_2">
<label>1.2</label>
<title>Emerging Datasets and Research Gaps</title>
<p>Two important datasets, NTU RGB&#x002B;D 120 and HMDB51, include videos of medical-related activities like sneezing, coughing, sitting, and walking, which are useful for healthcare. However, only a few researchers have used these datasets for MRHA recognition, and the reported accuracy is still low. Improving MRHA recognition using these datasets can lead to systems that are more accurate, private, and flexible. Future work should focus on creating better models that work well for different populations and activity types.</p>
</sec>
<sec id="s1_3">
<label>1.3</label>
<title>Research Motivation</title>
<p>The growing elderly population and increasing fall rates highlight the need for accurate, efficient, and privacy-preserving medical-related human activity (MRHA) recognition systems. Falls are a major cause of injury and mortality in older adults, driving up healthcare costs and affecting quality of life. Traditional solutions like wearable sensors and vision-based systems face issues such as low accuracy, portability challenges, and privacy concerns. This study proposes a robust solution leveraging IoT and mobile technology for real-time patient monitoring. By using skeleton data to capture joint movements while ensuring privacy, the research aims to develop an advanced human motion recognition (HMR) framework for healthcare. This approach enhances safety, improves outcomes, and reduces costs, advancing modern healthcare systems.</p>
</sec>
<sec id="s1_4">
<label>1.4</label>
<title>The Goal and Scope of the Study</title>
<p>This study develops a real-time Medical-Related Human Activity (MRHA) recognition system using a multi-stage deep learning model and IoT integration. It features direct mobile notifications without third-party apps for fast, secure health alerts. The system is rigorously validated for accuracy, offering timely, data-driven support for improved patient care. Key contributions includes:
<list list-type="bullet">
<list-item>
<p><bold>Novel Hybrid Deep Learning Model for MRHA Recognition:</bold> We propose ENConvLSTM, a multi-stage deep learning model combining EfficientNet for spatial features and ConvLSTM for spatio-temporal integration. It addresses key HMR challenges like high computational cost, low accuracy, and poor adaptability. Using seven MBConv blocks, it enhances spatial representation and motion analysis.</p></list-item>
<list-item>
<p><bold>Exceptional Performance on Benchmark Datasets:</bold> The model is evaluated on the NTU RGB&#x002B;D 120 that is presented in <xref ref-type="fig" rid="fig-1">Fig. 1</xref> and HMDB51 datasets, focusing on MRHA such as sneezing, falling, walking, and sitting. It achieves 94.85% accuracy for cross-subject evaluations and 96.45% for cross-view evaluations on NTU RGB&#x002B;D 120, along with 89.22% accuracy on HMDB51. These results demonstrate the model&#x2019;s capability to handle both spatial and temporal data aspects effectively.</p>
</list-item>
<list-item>
<p><bold>Real-Time IoT-Integrated MRHA Recognition System:</bold> A real-time IoT system using Raspberry Pi and a GSM module with Twilio API delivers instant SMS alerts, eliminating the need for third-party apps. It recognizes 12 MRHAs (e.g., sneezing, falling, walking), enabling early diagnosis, timely intervention, and improved healthcare outcomes.</p></list-item>
</list></p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Sample example of &#x201C;NTU RGB&#x002B;D 120&#x201D; dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_63563-fig-1.tif"/>
</fig>
<p>The paper is organized as follows: <xref ref-type="sec" rid="s2">Section 2</xref> presents the literature review and major contributions. <xref ref-type="sec" rid="s3">Section 3</xref> describes the dataset, selection criteria, and preprocessing. The proposed methodology is discussed in <xref ref-type="sec" rid="s4">Section 4</xref>, followed by experimental results and performance evaluation in <xref ref-type="sec" rid="s5">Section 5</xref>. Real-time implementation and analysis are presented in <xref ref-type="sec" rid="s6">Section 6</xref>. <xref ref-type="sec" rid="s7">Section 7</xref> concludes the paper and suggests future directions.</p>
</sec>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Works</title>
<p>The convergence of IoT and mobile technologies has revolutionized healthcare, enabling real-time patient monitoring and diagnosis. One of the crucial areas in this domain is human motion recognition (HMR), which various methodologies have employed over the last few years. In this survey, we consider only the &#x201C;NTU RGB&#x002B;D 120&#x201D; dataset. Cross-Subject (CS) and Cross-View (CV) are two accuracy evaluation methods. Plizzari et al. (2021) [<xref ref-type="bibr" rid="ref-12">12</xref>] proposed the Spatial-Temporal Transformer Network (ST-TR), combining a Graph Convolutional Network (GCN) and a transformer to enhance human activity recognition (HAR) by capturing global attention across spatial and temporal dimensions. The model achieved a cross-subject accuracy of 88.60% and a cross-view accuracy of 94.70%, effectively addressing spatial and temporal complexities in HAR. Ref. [<xref ref-type="bibr" rid="ref-13">13</xref>] introduced the Temporal-Aware Adaptive Skeleton Graph Network (TA-ASGN), which combines transformers with temporal adaptive skeleton graphs for improved HAR. The model achieved a cross-subject accuracy of 89.80% and a cross-view accuracy of 95.30%, excelling in subject variation and viewpoint adaptation. Duan et al. (2022) [<xref ref-type="bibr" rid="ref-14">14</xref>] developed a Graph Convolution and Transformer Hybrid Model for skeleton-based action recognition. By integrating GCNs for local spatial feature extraction and transformers for global temporal attention, the model achieved a cross-subject accuracy of 90.10% and a cross-view accuracy of 96.20%, demonstrating robustness in spatio-temporal action recognition. Zhao et al. (2022) [<xref ref-type="bibr" rid="ref-15">15</xref>] introduced Skeleton-Aware Geometry Feature Learning for HAR, which leverages geometric relationships in skeleton data to improve recognition accuracy. This method achieved a cross-subject accuracy of 90.00% and a cross-view accuracy of 95.80%, refining skeleton-based HAR through geometry-aware feature learning.</p>
<p>Ref. [<xref ref-type="bibr" rid="ref-16">16</xref>] introduced Temporal Edge Aggregation for GCN enhancing HAR by aggregating temporal edge information. This method achieved a cross-subject accuracy of 90.30% and a cross-view accuracy of 96.00%, improving the temporal modeling capabilities of GCN-based HAR systems. Ref. [<xref ref-type="bibr" rid="ref-17">17</xref>] proposed a hybrid model combining a GCN with a transformer to enhance skeleton-based HAR. By incorporating an attention mechanism, the model selectively focuses on important spatial-temporal features, achieving a cross-subject accuracy of 89.70% and a cross-view accuracy of 95.70%, demonstrating its effectiveness in capturing local and global dependencies. Ref. [<xref ref-type="bibr" rid="ref-18">18</xref>] introduced the Dual Stream Transformer GCN, a model integrating both temporal and spatial streams to capture dynamic temporal changes and spatial relationships. The model achieved a cross-subject accuracy of 91.00% and a cross-view accuracy of 96.50%, highlighting its superior performance in HAR tasks through effective combination of temporal and spatial features. Ref. [<xref ref-type="bibr" rid="ref-11">11</xref>] achieved over 94% accuracy on URFD and UPFD datasets using Graph Convolutional Networks (GCNs) combined with Convolutional Neural Networks (CNNs).</p>
<p>Moreover, existing systems are rarely evaluated on vision-based datasets designed explicitly for MRHA recognition beyond falls. Two promising datasets, NTU RGB&#x002B;D 120 and HMDB51, include medical-related human activity video data, offering new opportunities for MRHA recognition research. These datasets encompass a variety of medical activities, such as sneezing, coughing, sitting, and walking, which are relevant to healthcare scenarios. Despite this, few researchers have developed MRHA recognition systems based on these datasets, and their reported accuracy levels remain unsatisfactory. Addressing the limitations of existing approaches and leveraging these datasets for MRHA recognition could lead to the development of more effective, privacy-preserving, and versatile systems. Future work should focus on advancing vision-based MRHA recognition models, improving their accuracy and generalizability across diverse populations and activity types.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Dataset Description</title>
<p>This study leverages existing HMR datasets, prioritizing datasets based on three criteria: (1) inclusion of activities listed in <xref ref-type="table" rid="table-1">Table 1</xref>, (2) relevance to the medical domain, and (3) richness of features in the video dataset. Among the analyzed datasets, NTU RGB&#x002B;D 120 emerged as the most suitable due to its overlap with key activities and its potential for improvement, as highlighted in the literature <ext-link ext-link-type="uri" xlink:href="https://rose1.ntu.edu.sg/dataset/actionRecognition/">https://rose1.ntu.edu.sg/dataset/actionRecognition/</ext-link> (accessed on 28 April 2025).</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Human Motion Recognition (HMR) datasets with some class activities</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>S.N.</th>
<th>Dataset name</th>
<th>Total samples data</th>
<th>Total classes</th>
</tr>
</thead>
<tbody>
<tr>
<td>1</td>
<td>ActivityNet</td>
<td>21,313</td>
<td>200</td>
</tr>
<tr>
<td>2</td>
<td>Charades</td>
<td>66,493</td>
<td>157</td>
</tr>
<tr>
<td>3</td>
<td>HMDB51</td>
<td>6766</td>
<td>51</td>
</tr>
<tr>
<td>4</td>
<td>NTU RGB&#x002B;D 120</td>
<td>114,480</td>
<td>120</td>
</tr>
<tr>
<td>5</td>
<td>STAIR Actions</td>
<td>109,478</td>
<td>100</td>
</tr>
<tr>
<td>6</td>
<td>UCF101</td>
<td>13,320</td>
<td>101</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3_1">
<label>3.1</label>
<title>&#x201C;NTU RGB&#x002B;D 120&#x201D; Dataset</title>
<p>The NTU RGB&#x002B;D 120 dataset, developed by Nanyang Technological University, is a benchmark for human action recognition, featuring 120 motion classes and 114,480 samples captured in RGB video format (.mp4) at 1920 <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 1080 resolution with 24 fps. Each sample includes 3D coordinates of 25 body joints, enabling detailed skeletal motion analysis. For this study, 12 healthcare-related activities were selected, such as sneezing/cough, staggering, and chest pain, totaling 13,200 samples, divided into 80% training (10,560 samples) and 20% testing (2640 samples) splits.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>&#x201C;HMDB51&#x201D; Dataset</title>
<p>We conducted experiments on the HMDB51 dataset, a benchmark for human action recognition. The HMDB51 dataset comprises <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mn>6766</mml:mn></mml:math></inline-formula> video clips with a total file size of <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mn>2</mml:mn><mml:mtext>&#xA0;</mml:mtext><mml:mrow><mml:mi mathvariant="normal">G</mml:mi><mml:mi mathvariant="normal">B</mml:mi></mml:mrow></mml:math></inline-formula>. It features <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mn>51</mml:mn></mml:math></inline-formula> action categories, with each category containing at least 101 video clips sourced from diverse origins, including movies, YouTube, and other online platforms. For this study, we selected six classes: walk, stand, eat, sit, and drink&#x2014;particularly relevant to medical and healthcare applications. These classes were chosen due to their critical importance in healthcare scenarios where activity recognition can provide meaningful insights and enhance patient monitoring systems.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Proposed Method</title>
<p>There are many researchers who have been working to develop a human motion recognition (HMR) system. However, a few researchers have been working to develop IoT-integrating HMR systems. we propose a new HMR method that uses spatial and temporal features powered by multi-stage deep learning and integrated with IoT. First, we use EfficientNet to extract spatial features from skeleton frame sequences. EfficientNet is designed with seven Mobile Inverted Bottleneck Convolutions (MBConv) blocks. Each block includes a convolutional layer, a depthwise separable layer, and a squeeze-and-excitation (SE) module to create optimized feature representations. These spatial features are then passed to ConvLSTM, which extracts spatio-temporal features, combining spatial and sequential information. The main components of our study are given below:
<list list-type="bullet">
<list-item>
<p><bold>OpenPose Based BodyPose Extraction:</bold> We employed OpenPose to extract 25 key points from the whole for each frame in the sequence.</p></list-item>
<list-item>
<p><bold>Hybrid Deep Learning Architecture:</bold> We introduce a novel hybrid deep learning model, ENConvLSTM, combining EfficientNet and ConvLSTM to address challenges in HMR, such as high computational demands, low accuracy, and adaptability. The proposed architecture consists of two key components: efficientnet to extract the spatial feature from input skeleton data. It utilizes seven Mobile Inverted Bottleneck Convolutions (MBConv) blocks. Each MBConv block comprises a <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> convolution layer for feature mapping and a depthwise separable convolution layer for dimensionality reduction. A squeeze-and-excitation (SE) module for adaptive feature recalibration. Then we extracted the spatio-temporal feature within ConvLSTM. This takes the spatial features generated by EfficientNet and models the temporal dependencies across frames, producing robust spatio-temporal features.</p></list-item>
<list-item>
<p><bold>Classification Module:</bold> The spatio-temporal features generated by ConvLSTM are passed through a classification module, which includes Global Average Pooling and Fully Connected Layer, etc. This multi-stage pipeline ensures robust motion recognition by integrating spatial and temporal modeling techniques.</p></list-item>
<list-item>
<p><bold>IoT Integration for Real-Time Alerts:</bold> The system integrates IoT components, including a Raspberry Pi and GSM module, to provide real-time alerts. Twilio&#x2019;s SMS API is used to send instant notifications to caregivers and patients, removing the dependency on third-party mobile applications. This feature enhances the system&#x2019;s scalability and usability for healthcare scenarios.</p></list-item>
</list></p>
<sec id="s4_1">
<label>4.1</label>
<title>Data Preprocessing</title>
<p>The preprocessing and pose keypoint extraction process begins with detecting 25 skeletal joints of the human body using the Kinect v2 camera. These joints include key points such as the head, shoulders, elbows, wrists, hips, knees, ankles, and feet, each represented by 3D coordinates <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mo stretchy="false">(</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mi>Y</mml:mi><mml:mo>,</mml:mo><mml:mi>Z</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>. RGB videos are recorded at a resolution of <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mn>1920</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1080</mml:mn></mml:math></inline-formula>, while depth maps are captured at <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mn>512</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>424</mml:mn></mml:math></inline-formula> resolution. Skeletons are extracted from video frames at a rate of 24 frames per second (FPS), enabling precise tracking of motion trajectories. <xref ref-type="fig" rid="fig-2">Fig. 2</xref> illustrates the skeletal configuration, which abstracts human poses and movements while preserving spatial and temporal features crucial for motion analysis. The OpenPose library is used to extract skeletal data, identifying and tracking the 25 body joints for each individual in the scene. Each frame provides a reduced yet comprehensive skeletal representation of human movement, significantly simplifying raw video data. This abstraction captures essential motion patterns, allowing for the analysis of complex motion dynamics while reducing computational complexity. The skeleton data structure provides an efficient input format for deep learning models, focusing on critical movement patterns. The preprocessing pipeline further enhances the data for analysis. Video frames are sampled at 10 frames per second (FPS) to eliminate redundancy while retaining critical motion information. Each frame is resized to a uniform resolution, converted to grayscale, and normalized to ensure consistency and compatibility across samples. Finally, the processed frames are organized into sequential arrays to represent the temporal dynamics of motion and fed into the feature extraction module.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>The workflow of the proposed methodology this figure represents an end-to-end system for human motion detection using a deep learning model, followed by real-time monitoring and notification</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_63563-fig-2.tif"/>
</fig>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Spatial Temporal Feature Extraction</title>
<p>The sequential skeleton information is fed into the feature extraction module. Here first we extract the spatial feature using EfficientNet and then we fed him into ConvLSTM to extract the spatiotemporal features defined below.</p>
<sec id="s4_2_1">
<label>4.2.1</label>
<title>EfficientNet Model</title>
<p>EfficientNet [<xref ref-type="bibr" rid="ref-19">19</xref>] is a state-of-the-art deep learning model for spatial feature enhancement and image classification tasks. It achieves high accuracy with fewer parameters and lower computational costs by scaling depth, width, and resolution in a balanced manner due to the depthwise separable convolution and sequence excitation module. <xref ref-type="fig" rid="fig-3">Fig. 3b</xref> shows the efficient net model diagram, which was constructed with various deep learning modules to extract the spatial feature from input skeleton data. It utilizes seven Mobile Inverted Bottleneck Convolutions (MBConv) blocks. Each MBConv block comprises a 1 <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 1 convolution layer for feature mapping and a depthwise separable convolution layer for dimensionality reduction. A squeeze-and-excitation (SE) module for adaptive feature recalibration. The series of MBConv is mainly used to downsample and extract meaningful features from the input, which is demonstrated in <xref ref-type="fig" rid="fig-3">Fig. 3d</xref>. The structure consists of multiple blocks of convolutions, where the operations can be represented as:
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>;</mml:mo><mml:mi>W</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mtext>MBConv</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>;</mml:mo><mml:mi>W</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mi>x</mml:mi></mml:math></inline-formula> is the input skeleton features, <italic>W</italic> are the convolutional weights, and <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:mi>y</mml:mi></mml:math></inline-formula> is the output feature map. EfficientNet progressively reduces the spatial resolution while expanding the depth of features, leading to high-level representations for the next stage. These blocks apply depth-wise separable convolutions to efficiently extract hierarchical spatial features, resulting in feature maps from progressively lower resolutions and deeper feature representations.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title><bold>(a)</bold> Proposed multi-stage deep learning model constructed with <bold>(b)</bold> EfficientNet and <bold>(c)</bold> ConvLSTM beside the classification module <bold>(d)</bold> Mobile Inverted Bottleneck Convolutions [<xref ref-type="bibr" rid="ref-20">20</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_63563-fig-3.tif"/>
</fig>
</sec>
<sec id="s4_2_2">
<label>4.2.2</label>
<title>ConvLSTM Model</title>
<p>Then we extracted the spatio-temporal feature within ConvLSTM [<xref ref-type="bibr" rid="ref-21">21</xref>]. This takes the spatial features generated by EfficientNet and models the temporal dependencies across frames, producing robust spatio-temporal features. This hybrid approach addresses the limitations of existing methods, such as high computational complexity, low accuracy, and limited adaptability to diverse healthcare scenarios. <xref ref-type="fig" rid="fig-3">Fig. 3c</xref> demonstrated the ConvLSTM network diagram, which mainly extended the capabilities of traditional LSTM networks by integrating convolutional layers.</p>
</sec>
<sec id="s4_2_3">
<label>4.2.3</label>
<title>ENConvLSTM Proposed Model Architecture</title>
<p><xref ref-type="fig" rid="fig-3">Fig. 3</xref> presents a hybrid architecture for human motion recognition, combining EfficientNet for spatial feature extraction and ConvLSTM for temporal feature modeling. At the top, the EfficientNet architecture is shown, which processes input frames (224 <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 224 resolution) through a series of MBConv blocks. Once skeleton data is extracted, it is passed through the EfficientNet model for spatial feature extraction.</p>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Experiment and Result</title>
<p>The proposed model is evaluated on the NTU RGB&#x002B;D 120 dataset, focusing on 12 selected medical classes. The dataset is divided into training (80%) and testing (20%) portions. The model is trained on 80% of the data, with its performance evaluated on the remaining 20%. During training, the ENConvLSTM model minimizes classification loss, typically cross-entropy, using the Adam optimizer to learn and differentiate between various human activities. Key performance metrics accuracy, precision, recall, and F1-score are calculated for both cross-subject and cross-view evaluations [<xref ref-type="bibr" rid="ref-22">22</xref>&#x2013;<xref ref-type="bibr" rid="ref-24">24</xref>]. The model performs exceptionally well, particularly in cross-view evaluations, which involve more significant variability due to different camera angles, demonstrating its robustness and generalizability for real-time Human Motion Recognition (HMR) in healthcare applications.</p>
<sec id="s5_1">
<label>5.1</label>
<title>Software and Hardware Requirements</title>
<p>The study requires several key software and hardware components. The software includes deep learning frameworks like TensorFlow, PyTorch, scikit-learn, and Keras, with data processing libraries such as NumPy, Pandas, and OpenCV, all using Python. Development is done in Jupyter Notebook and PyCharm. The hardware setup features an AMD Ryzen 9 5900X 12-Core Processor, running on a 64-bit system with Python 3.9.13 and CUDA 11.0. It includes an NVIDIA&#x00AE; GeForce RTX 3060 graphics card with 6 GB of memory, 64 GB of RAM, and a 4 TB SSD for storage. An Arduino UNO REV3 and a SIM900A GSM module are also used for SMS communication.</p>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Ablation Study</title>
<p>The ablation study, as shown in <xref ref-type="table" rid="table-2">Table 2</xref>, compares the performance of the proposed model with several baseline methods on the HMDB51 dataset using accuracy, precision, recall, and F1-score. It shows that the proposed combination achieves high performance accuracy compared to the baseline individual methods.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Ablation study of the proposed model with HMDB51 dataset</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Methods</th>
<th>Accuracy (%)</th>
<th>Precision (%)</th>
<th>Recall (%)</th>
<th>F1-Score (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td>LSTM</td>
<td>76.44</td>
<td>75.96</td>
<td>76.42</td>
<td>78.18</td>
</tr>
<tr>
<td>ConvLSTM</td>
<td>70.35</td>
<td>71.07</td>
<td>73.33</td>
<td>69.28</td>
</tr>
<tr>
<td>EfficientNetB0</td>
<td>72.57</td>
<td>74.24</td>
<td>72.57</td>
<td>75.45</td>
</tr>
<tr>
<td>EfficientNet (B0&#x2013;B7)</td>
<td>76.75</td>
<td>74.48</td>
<td>71.73</td>
<td>73.65</td>
</tr>
<tr>
<td><bold>Proposed (ENConvLSTM)</bold></td>
<td><bold>89.22</bold></td>
<td>88.12</td>
<td>86.54</td>
<td>87.96</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5_3">
<label>5.3</label>
<title>Proposed Model Parameters List</title>
<p>The proposed model has 8,138,104 parameters, of which 8,095,640 are trainable and 42,464 are non-trainable. The model is trained with a batch size of 16 for 100 epochs using sparse categorical cross-entropy as the loss function and Adam as the optimizer, with a learning rate of 0.001. <xref ref-type="fig" rid="fig-4">Fig. 4</xref> highlights Adam&#x2019;s superior performance over SGD, RMSprop, and Adagrad in training the proposed model. Adam achieves the highest validation accuracy (approaching 0.95) and the lowest validation loss due to its adaptive learning rate mechanism, which accelerates convergence and improves generalization. The smooth accuracy and loss curves further demonstrate Adam&#x2019;s stability and effectiveness, whereas other optimizers show slower convergence and higher validation losses. These results confirm Adam as the most effective optimizer for this model.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Model optimizers comparison: <bold>(a)</bold> Validation accuracy and <bold>(b)</bold> Validation loss curve for NTU dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_63563-fig-4.tif"/>
</fig>
</sec>
<sec id="s5_4">
<label>5.4</label>
<title>Performance Matrix with NTU RGB&#x002B;D 120 Dataset</title>
<p>We evaluated the proposed method with the NTU RGB&#x002B;D medical related action dataset with various configurations, including cross-subject evaluation and cross-view evaluation. It involves splitting the dataset such that training and testing samples are from different subjects, ensuring the model generalizes well to unseen individuals. The model achieved a cross-subject accuracy of 94.85%. It involves splitting the dataset based on different camera views, ensuring the model performs well across various perspectives. The model achieved a cross-view accuracy of 96.45%. Key performance metrics&#x2014;accuracy, precision, recall, and F1-score&#x2014;are calculated for both cross-subject and cross-view evaluations, as shown in <xref ref-type="table" rid="table-3">Table 3</xref>.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Other performance evaluation metrics on NTU RGB&#x002B;D 120 dataset</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>S.N.</th>
<th>Performance martics</th>
<th colspan="2">Dataset evaluation methods</th>
</tr>
<tr>
<th></th>
<th></th>
<th>Cross-Subject (%)</th>
<th>Cross-View (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td>1</td>
<td>Accuracy</td>
<td>94.85</td>
<td>96.45</td>
</tr>
<tr>
<td>2</td>
<td>Precision</td>
<td>93.70</td>
<td>95.90</td>
</tr>
<tr>
<td>3</td>
<td>Recall</td>
<td>94.30</td>
<td>96.10</td>
</tr>
<tr>
<td>4</td>
<td>F1-Score</td>
<td>94.00</td>
<td>96.00</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In the cross-subject evaluation, the accuracy exhibits a robust upward trend from an initial 75.01% to a final 96.85%, with some minor fluctuations and eventual stabilization. Concurrently, the loss decreases consistently from 0.60 to 0.04, reflecting an overall improvement in model performance. Similarly, the cross-view evaluation shows a strong accuracy progression, starting at 78.34% and reaching 96.45% by the end, with minor variations throughout and a stabilizing trend. The loss trend in this evaluation mirrors that of the cross-subject assessment, declining from 0.55 to 0.03, indicating effective learning and convergence. Both evaluations demonstrate an effective model performance enhancement over time, with accuracy improving and loss decreasing, culminating in stable performance metrics.</p>
<sec id="s5_4_1">
<label>5.4.1</label>
<title>State of the Art Comparison for the NTU RGB&#x002B;D 120 Dataset</title>
<p>The performance of the ENConvLSTM model is compared with several state-of-the-art models, including traditional methods and recent deep-learning approaches shows in <xref ref-type="table" rid="table-4">Table 4</xref>. Each Performance Evaluation Method (Cross-View and Cross-Subject) is graphically presented in <xref ref-type="fig" rid="fig-5">Fig. 5a</xref>,<xref ref-type="fig" rid="fig-5">b</xref>. The proposed ENConvLSTM model significantly outperforms the other models across cross-subject and cross-view evaluations on the NTU RGB&#x002B;D 120 dataset, showing excellent accuracy, precision, recall, and F1 score results. This indicates that the proposed model is highly robust and performs well across different subjects and viewing conditions.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>State of the art comparison for NTU RGB&#x002B;D 120 dataset</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Used model with citation</th>
<th colspan="2">NTU RGB&#x002B;D 120</th>
</tr>
<tr>
<th>Methods</th>
<th>Cross-Subject (%)</th>
<th>Cross-View (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td>MS-G3D [<xref ref-type="bibr" rid="ref-25">25</xref>]</td>
<td>86.92</td>
<td>88.44</td>
</tr>
<tr>
<td>PA-ResGCN-B19 [<xref ref-type="bibr" rid="ref-26">26</xref>]</td>
<td>87.34</td>
<td>88.37</td>
</tr>
<tr>
<td>Dynamic GCN [<xref ref-type="bibr" rid="ref-27">27</xref>]</td>
<td>87.37</td>
<td>88.68</td>
</tr>
<tr>
<td>CTR-GCN [<xref ref-type="bibr" rid="ref-28">28</xref>]</td>
<td>88.99</td>
<td>90.62</td>
</tr>
<tr>
<td>4s Shift-GCN [<xref ref-type="bibr" rid="ref-29">29</xref>]</td>
<td>85.9</td>
<td>87.6</td>
</tr>
<tr>
<td>ST-TR [<xref ref-type="bibr" rid="ref-30">30</xref>]</td>
<td>89.95</td>
<td>96.12</td>
</tr>
<tr>
<td>GA-GCN [<xref ref-type="bibr" rid="ref-31">31</xref>]</td>
<td>92.38</td>
<td>92.83</td>
</tr>
<tr>
<td><bold>ENConvLSTM (Proposed)</bold></td>
<td><bold>94.85</bold></td>
<td><bold>96.45</bold></td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Performance comparison among different tested models on the NTU RGB&#x002B;D 120 dataset with <bold>(a)</bold> cross-view <bold>(b)</bold> cross-subject configuration</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_63563-fig-5.tif"/>
</fig>
</sec>
</sec>
<sec id="s5_5">
<label>5.5</label>
<title>Performance Accuracy and State of the Art Comparison for HMDB51 Dataset</title>
<p><xref ref-type="fig" rid="fig-6">Fig. 6</xref> presents the confusion matrix and loss with various optimizations. Therefore, we consider the Adam optimizer for this study. This matrix offers a detailed view of accuracy for each individual class. <xref ref-type="fig" rid="fig-6">Fig. 6a</xref> shows the confusion matrix of the proposed model for the HMDB15 dataset. From this matrix, we can see that the diagonal line represents the true classes, while the off-diagonal data represents false detections. The classes &#x201C;drink,&#x201D; &#x201C;eat,&#x201D; and &#x201C;walk&#x201D; achieve remarkable accuracy in this dataset. However, the other classes show comparatively lower accuracy. We analyze the validation loss curve relative to the number of iterations, as illustrated in <xref ref-type="fig" rid="fig-6">Fig. 6b</xref>. The curve indicates that the ADAM optimizer outperforms both the SGD and RMSProp algorithms. Based on the analysis, Adam is likely the best-performing optimizer among the three optimizers, such as Adam [<xref ref-type="bibr" rid="ref-32">32</xref>], SGD [<xref ref-type="bibr" rid="ref-33">33</xref>,<xref ref-type="bibr" rid="ref-34">34</xref>], and RMSProp [<xref ref-type="bibr" rid="ref-33">33</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>]. This is because Adam combines the advantages of both RMSProp and momentum, leading to faster and more reliable convergence [<xref ref-type="bibr" rid="ref-33">33</xref>]. It adapts the learning rate during training and uses past gradient information to accelerate learning, making it efficient and effective for many deep learning problems [<xref ref-type="bibr" rid="ref-35">35</xref>]. These standard classification metrics for the HMDB51 dataset are listed in <xref ref-type="table" rid="table-5">Table 5</xref>, where the average accuracy in our experiment is 89.22%. <xref ref-type="table" rid="table-6">Table 6</xref> provides a comparative analysis of various models tested on the HMDB51 dataset for human action recognition. The proposed model, a multi-stage model, achieves the highest accuracy of 89.22%, showcasing its superior performance. Models such as STM Framework [<xref ref-type="bibr" rid="ref-36">36</xref>] and Attention-Based LSTM with 3D CNN [<xref ref-type="bibr" rid="ref-37">37</xref>] demonstrate strong performance with accuracies of 80.40% and 87.98%, respectively, while EfficientNet delivers an impressive accuracy of 88.70% [<xref ref-type="bibr" rid="ref-38">38</xref>]. In contrast, earlier models like VicTR (B/16) [<xref ref-type="bibr" rid="ref-39">39</xref>] and SVT Self-Supervised Transfer [<xref ref-type="bibr" rid="ref-40">40</xref>] achieve relatively lower accuracies of 67.28% and 57.80%, respectively. Additionally, reference [<xref ref-type="bibr" rid="ref-41">41</xref>] proposed a Vision Transformer (ViT) model, achieving 59.74% and 68.2% accuracy on HMDB51 by leveraging self-attention. Moreover, reference [<xref ref-type="bibr" rid="ref-42">42</xref>] introduced a Dual-Stream Framework, achieving 78.62% accuracy by separately processing temporal and spatial features for improved action recognition, surpassing recent approaches over traditional methodologies.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title><bold>(a)</bold> Confusion matrix <bold>(b)</bold> Loss curve for the HMDB51 dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_63563-fig-6.tif"/>
</fig><table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Classification result for the HMDB51 dataset</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Selected class levels</th>
<th>Precision (%)</th>
<th>Recall (%)</th>
<th>F1-Score (%)</th>
<th>Accuracy (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td>drink</td>
<td>89.00</td>
<td>93.00</td>
<td>91.00</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>eat</td>
<td>89.00</td>
<td>95.00</td>
<td>92.00</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>fall_floor</td>
<td>84.00</td>
<td>69.00</td>
<td>76.00</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>sit</td>
<td>85.00</td>
<td>86.00</td>
<td>86.00</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>stand</td>
<td>91.00</td>
<td>79.00</td>
<td>85.00</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>walk</td>
<td>91.00</td>
<td>96.00</td>
<td>93.00</td>
<td>&#x2013;</td>
</tr>
<tr>
<td><bold>Average</bold></td>
<td>88.17</td>
<td>86.33</td>
<td>87.17</td>
<td><bold>89.22</bold></td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-6">
<label>Table 6</label>
<caption>
<title>State-of-the-art comparison for the proposed model with HMDB51 dataset (Sort by Year)</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Author and citation</th>
<th>Model name</th>
<th>Year</th>
<th>Accuracy (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td>Ranasinghe et al. [<xref ref-type="bibr" rid="ref-40">40</xref>]</td>
<td>Self-supervised transfer</td>
<td>2022</td>
<td>67.28, 57.8</td>
</tr>
<tr>
<td>Sarraf et al. [<xref ref-type="bibr" rid="ref-41">41</xref>]</td>
<td>ViT</td>
<td>2023</td>
<td>59.74, 68.2</td>
</tr>
<tr>
<td>Saoudi et al. [<xref ref-type="bibr" rid="ref-37">37</xref>]</td>
<td>Attention-LSTM-3DCNN</td>
<td>2023</td>
<td>87.98</td>
</tr>
<tr>
<td>Burton-Barr et al. [<xref ref-type="bibr" rid="ref-38">38</xref>]</td>
<td>Act-control vision</td>
<td>2024</td>
<td>88.70</td>
</tr>
<tr>
<td>Kahatapitiya et al. [<xref ref-type="bibr" rid="ref-39">39</xref>]</td>
<td>VicTR (B/16)</td>
<td>2024</td>
<td>51.00</td>
</tr>
<tr>
<td>Hussain et al. [<xref ref-type="bibr" rid="ref-43">43</xref>]</td>
<td>Dual-stream framework</td>
<td>2024</td>
<td>78.62</td>
</tr>
<tr>
<td>Jiang et al. [<xref ref-type="bibr" rid="ref-36">36</xref>]</td>
<td>STM framework</td>
<td>2024</td>
<td>80.40</td>
</tr>
<tr>
<td><bold>Proposed</bold></td>
<td><bold>EfficientNetB0ConvLSTM</bold></td>
<td><bold>2024</bold></td>
<td><bold>89.22</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s6">
<label>6</label>
<title>Real-Deployment</title>
<p>Finally, the proposed deep learning system is integrated with a Raspberry Pi and GSM module to send real-time alerts via Twilio SMS service, keeping caregivers and patients informed instantly. This system is scalable, efficient, and proactive, helping to improve patient monitoring and outcomes and reduce healthcare costs.</p>
<sec id="s6_1">
<label>6.1</label>
<title>Real Time Implementation</title>
<p><xref ref-type="fig" rid="fig-7">Fig. 7</xref> shows the interconnection between an Arduino UNO and the SIM900 GSM module to implement the real-time scenario. A 12V 2Amp DC adapter powers the Arduino. The TX (transmit) and RX (receive) pins are crucial for communication. The TX pin of the Arduino connects to the RXD pin of the TTL module for data transmission, while the RX pin of the Arduino connects to the RDX pin of the TTL module for receiving data. Both devices share a common ground (GND) for proper operation. This setup enables the Arduino to communicate with the GSM module for tasks like sending SMS or making voice calls. The connections include power and ground lines: the red wire represents the VCC (power) connection from the Raspberry Pi to the ESP8266, while the blue wire indicates the ground (GND) connection. The yellow wire connects the GPIO pin from the Raspberry Pi to the ESP8266&#x2019;s TX pin for data transmission, and the green wire connects another GPIO pin to the RX pin of the ESP8266 for data reception. This setup enables the Raspberry Pi to communicate with the ESP8266 for wireless connectivity in various projects, such as IoT applications. Arduino UNO REV3-Compact and Versatile Microcontroller Board with A000066. Interface a SIM900A GSM module with an Arduino to send and receive SMS. Arduino UNO REV3-Compact and Versatile Microcontroller Board with A000066. Interface a SIM900A GSM module with an Arduino to send and receive SMS. <xref ref-type="fig" rid="fig-3">Fig. 3</xref> illustrates the proposed system for human motion recognition using an ENConvLSTM model integrated with an Arduino. It begins with an input video that is processed to extract skeleton data, represented in a 3D coordinate system (X, Y, Z). This skeleton data is fed into the ENConvLSTM model, which then utilizes a SoftMax layer to predict the activity being performed. The predicted activity is communicated to an Arduino, which can be powered by a stable 5V source or battery, indicating a real-time or recorded data application. This setup enables effective monitoring and recognition of human activities through skeletal motion analysis. The proposed human motion recognition system&#x2019;s performance in the laboratory experiment is satisfactory. However, there are always some differences between laboratory and real-time scenarios. <xref ref-type="table" rid="table-7">Table 7</xref> illustrates the result in a real-time scenario, and its visual representation is shown in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Schematic diagram illustrating the connections between an Arduino UNO and a SIM900 GSM TTL module, highlighting the power supply, TX and RX pin connections, and the common ground to facilitate communication for tasks such as SMS sending and voice calling</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_63563-fig-7.tif"/>
</fig><table-wrap id="table-7">
<label>Table 7</label>
<caption>
<title>Laboratory and real-time experiment performance for the HMDB51 dataset</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="center">No.</th>
<th align="center">Actual class</th>
<th align="center">Real-time observation (individual)</th>
<th align="center">Laboratory experiment (%)</th>
<th align="center">Real-time experiment (%)</th>
<th align="center">Calculate score (%)</th>
<th align="center">Loss of real-time experiment (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td>1</td>
<td>fall_floor</td>
<td>Individual fall floor</td>
<td>89.00</td>
<td>87.00</td>
<td>97.75</td>
<td>2.00</td>
</tr>
<tr>
<td>2</td>
<td>walk</td>
<td>Individual walking</td>
<td>95.00</td>
<td>91.00</td>
<td>95.79</td>
<td>4.00</td>
</tr>
<tr>
<td>3</td>
<td>stand</td>
<td>Individual standing</td>
<td>79.00</td>
<td>77.00</td>
<td>97.47</td>
<td>2.00</td>
</tr>
<tr>
<td>4</td>
<td>eat</td>
<td>Individual eating</td>
<td>94.00</td>
<td>90.00</td>
<td>95.74</td>
<td>4.00</td>
</tr>
<tr>
<td>5</td>
<td>sit</td>
<td>Individual sitting</td>
<td>86.00</td>
<td>82.00</td>
<td>95.35</td>
<td>4.00</td>
</tr>
<tr>
<td>6</td>
<td>drink</td>
<td>Individual drinking</td>
<td>92.00</td>
<td>88.00</td>
<td>95.65</td>
<td>4.00</td>
</tr>
<tr>
<td colspan="3"><bold>Average score</bold></td>
<td><bold>89.22</bold></td>
<td><bold>85.83</bold></td>
<td><bold>96.29</bold></td>
<td><bold>3.33</bold></td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Real-Time human motion recognition performance metrics (Includes Accuracy and Loss Values) [NTU RGB&#x002B;D 120 dataset]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_63563-fig-8.tif"/>
</fig>
</sec>
<sec id="s6_2">
<label>6.2</label>
<title>Sending Alert Message (Notification)</title>
<p>In our proposed system, the model predicts human actions, specifically identifying whether an individual is experiencing a fall. Upon detecting a fall event, the system promptly triggers an alert message to the registered user mobile device. This real-time communication is facilitated through the integration of Twilio [<xref ref-type="bibr" rid="ref-44">44</xref>], a leading SMS and communication server provider, ensuring reliable and instantaneous notifications. By leveraging Twilio&#x2019;s robust platform, we enhance the responsiveness of our human motion recognition system, thereby significantly improving user safety and emergency response efficiency.</p>
</sec>
<sec id="s6_3">
<label>6.3</label>
<title>Work Limitations</title>
<p>While our paper highlights the integration of IoT components like Raspberry Pi and the GSM module, we acknowledge limitations in latency, reliability, and scalability in real-time scenarios. The system is not optimized for handling multiple alerts through efficient data transmission and prioritization, which could be explored in future work. Moreover, challenges such as imbalanced class distributions and noise in real-world data may also affect model performance. Future research could explore differential privacy and federated learning for enhanced security, particularly in the context of sensitive patient data.</p>
</sec>
</sec>
<sec id="s7">
<label>7</label>
<title>Conclusion and Future Work</title>
<p>This study presents a novel IoT-based framework for real-time Medical-Related Human Activity (MRHA) recognition, addressing challenges like high computational demands and low accuracy in existing systems. By combining EfficientNet for spatial feature extraction and ConvLSTM for spatio-temporal integration, the method achieves strong performance, with 94.85% accuracy for cross-subject and 96.45% for cross-view evaluations on the NTU RGB&#x002B;D 120 dataset. It also demonstrates 89.00% accuracy on the HMDB51 dataset. A key contribution is integrating the MRHA system with a Raspberry Pi and GSM module, providing real-time alerts through SMS. The system shows promise for improving patient monitoring and healthcare outcomes, although it may face challenges in real-time applications due to environmental factors. Future work will focus on multimodal datasets, cloud computing for remote monitoring, and addressing privacy concerns, ensuring the system practical use in healthcare settings.</p>
</sec>
</body>
<back>
<ack>
<p>I would like to express my sincere gratitude to the ICT Division of the Ministry of Posts, Telecommunications, and Information Technology of the People&#x2019;s Republic of Bangladesh for their invaluable support.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>This research was funded by the ICT Division of the Ministry of Posts, Telecommunications, and Information Technology of Bangladesh under Grant Number 56.00.0000.052.33.005.21-7 (Tracking No. 22FS15306), with support from the University of Rajshahi.</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>Md. Ekramul Hamid led the research and supervised the project. Subrata Kumer Paul and Abu Saleh Musa Miah designed the methodology; Subrata Kumer Paul also collected and processed data. Software was implemented by Rakhi Rani Paul, Subrata Kumer Paul, and Md. Ekramul Hamid. The draft was written by Subrata Kumer Paul, Abu Saleh Musa Miah, and Rakhi Rani Paul, with all authors contributing to revisions. Jungpil Shin handled administration and funding, while Md Abdur Rahim oversaw validation and review. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>NTU RGB&#x002B;D 120 Dataset: <ext-link ext-link-type="uri" xlink:href="https://rose1.ntu.edu.sg/dataset/actionRecognition/">https://rose1.ntu.edu.sg/dataset/actionRecognition/</ext-link> (accessed on 28 April 2025); Github Source Code: <ext-link ext-link-type="uri" xlink:href="https://www.github.com/Subrata11/IoT-based-Real-time-Human-Motion-Recognition-Based-on-Skeletons-">www.github.com/Subrata11/IoT-based-Real-time-Human-Motion-Recognition-Based-on-Skeletons-</ext-link> (accessed on 28 April 2025).</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lord</surname> <given-names>SR</given-names></string-name></person-group>. <article-title>Visual risk factors for falls in older people</article-title>. <source>Age Ageing</source>. <year>2006</year>;<volume>35</volume>(<issue>Suppl 2</issue>):<fpage>ii42</fpage>&#x2013;<lpage>5</lpage>. doi:<pub-id pub-id-type="doi">10.1093/ageing/afl085</pub-id>; <pub-id pub-id-type="pmid">16926203</pub-id></mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Egawa</surname> <given-names>R</given-names></string-name>, <string-name><surname>Miah</surname> <given-names>ASM</given-names></string-name>, <string-name><surname>Hirooka</surname> <given-names>K</given-names></string-name>, <string-name><surname>Tomioka</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Shin</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Dynamic fall detection using graph-based spatial temporal convolution and attention network</article-title>. <source>Electronics</source>. <year>2023</year>;<volume>12</volume>(<issue>15</issue>):<fpage>3234</fpage>. doi:<pub-id pub-id-type="doi">10.3390/electronics12153234</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hassan</surname> <given-names>N</given-names></string-name>, <string-name><surname>Miah</surname> <given-names>ASM</given-names></string-name>, <string-name><surname>Shin</surname> <given-names>J</given-names></string-name></person-group>. <article-title>A deep bidirectional LSTM model enhanced by transfer-learning-based feature extraction for dynamic human activity recognition</article-title>. <source>Appl Sci</source>. <year>2024</year>;<volume>14</volume>(<issue>2</issue>):<fpage>603</fpage>. doi:<pub-id pub-id-type="doi">10.3390/app14020603</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Hassan</surname> <given-names>N</given-names></string-name>, <string-name><surname>Miah</surname> <given-names>ASM</given-names></string-name>, <string-name><surname>Shin</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Enhancing human action recognition in videos through dense-level features extraction and optimized long short-term memory</article-title>. In: <conf-name>2024 7th International Conference on Electronics, Communications, and Control Engineering (ICECC)</conf-name>; <year>2024 Mar 22&#x2013;24</year>; <publisher-loc>Kuala Lumpur, Malaysia</publisher-loc>. <publisher-name>IEEE</publisher-name>; <volume>2024</volume>. p. <fpage>19</fpage>&#x2013;<lpage>23</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICECC63398.2024.00011</pub-id>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Romeo</surname> <given-names>L</given-names></string-name>, <string-name><surname>Marani</surname> <given-names>R</given-names></string-name>, <string-name><surname>Petitti</surname> <given-names>A</given-names></string-name>, <string-name><surname>Milella</surname> <given-names>A</given-names></string-name>, <string-name><surname>D&#x2019;Orazio</surname> <given-names>T</given-names></string-name>, <string-name><surname>Cicirelli</surname> <given-names>G</given-names></string-name></person-group>. <chapter-title>Image-based mobility assessment in elderly people from low-cost systems of cameras: a skeletal dataset for experimental evaluations</chapter-title>. In: <source>Ad-hoc, mobile, and wireless networks</source>. <publisher-loc>Cham</publisher-loc>. <publisher-name>Springer</publisher-name>; <year>2020</year>. p. <fpage>125</fpage>&#x2013;<lpage>30</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-030-61746-2_10</pub-id>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Paul</surname> <given-names>SK</given-names></string-name>, <string-name><surname>Miah</surname> <given-names>ASM</given-names></string-name>, <string-name><surname>Paul</surname> <given-names>RR</given-names></string-name>, <string-name><surname>Hamid</surname> <given-names>ME</given-names></string-name>, <string-name><surname>Shin</surname> <given-names>J</given-names></string-name>, <string-name><surname>Rahim</surname> <given-names>MA</given-names></string-name></person-group>. <article-title>IoT-based real-time medical-related human activity recognition using skeletons and multi-stage deep learning for healthcare</article-title>. arXiv:2501.07039. <year>2025</year>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Guti&#x00E9;rrez</surname> <given-names>J</given-names></string-name>, <string-name><surname>Rodr&#x00ED;guez</surname> <given-names>V</given-names></string-name>, <string-name><surname>Martin</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Comprehensive review of vision-based fall detection systems</article-title>. <source>Sensors</source>. <year>2021</year>;<volume>21</volume>(<issue>3</issue>):<fpage>947</fpage>. doi:<pub-id pub-id-type="doi">10.3390/s21030947</pub-id>; <pub-id pub-id-type="pmid">33535373</pub-id></mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Huang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Fang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Horn</surname> <given-names>BKP</given-names></string-name></person-group>. <article-title>Video-based fall detection for seniors with human pose estimation</article-title>. In: <conf-name>2018 4th International Conference on Universal Village (UV)</conf-name>; <year>2018 Oct 21&#x2013;24</year>; <publisher-loc>Boston, MA, USA</publisher-loc>. <publisher-name>IEEE</publisher-name>; <volume>2018</volume>. p. <fpage>1</fpage>&#x2013;<lpage>4</lpage>. doi:<pub-id pub-id-type="doi">10.1109/UV.2018.8642130</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>W</given-names></string-name>, <string-name><surname>Xiong</surname> <given-names>S</given-names></string-name></person-group>. <article-title>A real-time skeleton-based fall detection algorithm based on temporal convolutional networks and transformer encoder</article-title>. <source>Pervasive Mob Comput</source>. <year>2025</year>;<volume>107</volume>:<fpage>102016</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.pmcj.2025.102016</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Akash</surname> <given-names>HS</given-names></string-name>, <string-name><surname>Rahim</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Miah</surname> <given-names>ASM</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>HS</given-names></string-name>, <string-name><surname>Jang</surname> <given-names>SW</given-names></string-name>, <string-name><surname>Shin</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Two-stream modality-based deep learning approach for enhanced two-person human interaction recognition in videos</article-title>. <source>Sensors</source>. <year>2024</year>;<volume>24</volume>(<issue>21</issue>):<fpage>7077</fpage>. doi:<pub-id pub-id-type="doi">10.3390/s24217077</pub-id>; <pub-id pub-id-type="pmid">39517974</pub-id></mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zahan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Hassan</surname> <given-names>GM</given-names></string-name>, <string-name><surname>Mian</surname> <given-names>A</given-names></string-name></person-group>. <article-title>SDFA: structure-aware discriminative feature aggregation for efficient human fall detection in video</article-title>. <source>IEEE Trans Ind Inform</source>. <year>2023</year>;<volume>19</volume>(<issue>8</issue>):<fpage>8713</fpage>&#x2013;<lpage>21</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TII.2022.3221208</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Plizzari</surname> <given-names>C</given-names></string-name>, <string-name><surname>Cannici</surname> <given-names>M</given-names></string-name>, <string-name><surname>Matteucci</surname> <given-names>M</given-names></string-name></person-group>. <chapter-title>Spatial temporal transformer network for skeleton-based action recognition</chapter-title>. In: <source>ICPR International Workshops and Challenges</source>. <publisher-loc>Cham</publisher-loc>. <publisher-name>Springer</publisher-name>; <year>2021</year>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Shi</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Cheng</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Two-Stream Adaptive Graph Convolutional Networks for Skeleton-Based Action Recognition</article-title>. In: <conf-name> Proceedings of the IEEE/CVP Conferenceon Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2019</year>. p. <fpage>12026</fpage>&#x2013;<lpage>35</lpage>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Duan</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>K</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>D</given-names></string-name></person-group>. <article-title>DG-STGCN: dynamic spatial-temporal modeling for skeleton-based action recognition</article-title>. <comment>arXiv:2210.05895.2022</comment>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Usama</surname> <given-names>MM</given-names></string-name>, <string-name><surname>Ali</surname> <given-names>H</given-names></string-name>, <string-name><surname>Ahmed</surname> <given-names>J</given-names></string-name>, <string-name><surname>Ashraf</surname> <given-names>R</given-names></string-name>, <string-name><surname>Mahmood</surname><given-names>S</given-names></string-name> <string-name><surname>Ahmad</surname> <given-names>J</given-names></string-name> </person-group>. <article-title>Learning skeleton aware geometry features for action recognition</article-title>. <source>ST-RTR: spatial temporal relative transformer for skeleton-based human action recognition</source>. <comment>arXiv:2410.23806</comment>. <year>2024</year>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Heidari</surname> <given-names>N</given-names></string-name>, <string-name><surname>Iosifidis</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Temporal attention-augmented graph convolutional network for efficient skeleton-based human action recognition</article-title>. In: <conf-name>2020 25th International Conference on Pattern Recognition (ICPR)</conf-name>. <publisher-loc>Milan, Italy</publisher-loc>; <year>2021</year>. p. <fpage>7907</fpage>&#x2013;<lpage>14</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICPR48806.2021.9412091</pub-id>. </mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Cai</surname> <given-names>J</given-names></string-name></person-group>. <article-title> HybridNet: integrating GCN and CNN for skeleton-based action recognition</article-title>. <source>Appl Intell</source>. <year>2023</year>;<volume>53</volume>:<fpage>574</fpage>&#x2013;<lpage>85</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10489-022-03436-0</pub-id>. </mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>D</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>M</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Two-stream spatio-temporal GCN-transformer networks for skeleton-based action recognition</article-title>.<source> Sci Rep</source>. <year>2024</year>;<volume>15</volume>:<fpage>4982</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-025-87752-8</pub-id>; <pub-id pub-id-type="pmid">39929951</pub-id></mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Tan</surname> <given-names>M</given-names></string-name>, <string-name><surname>Le</surname> <given-names>QV</given-names></string-name></person-group>. <article-title>EfficientNet: rethinking model scaling for convolutional neural networks</article-title>. <comment>arXiv:1905.11946. 2019</comment>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Sandler</surname> <given-names>M</given-names></string-name>, <string-name><surname>Howard</surname> <given-names>A</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhmoginov</surname> <given-names>A</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>LC</given-names></string-name></person-group>. <article-title>MobileNetV2: inverted residuals and linear bottlenecks</article-title>. In: <conf-name>IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>; <year>2018 Jun 18&#x2013;23</year>; <publisher-loc>Salt Lake City, UT, USA</publisher-loc>. <publisher-name>IEEE; 2018</publisher-name>. p. <fpage>4510</fpage>&#x2013;<lpage>20</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2018.00474</pub-id>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>R</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>F</given-names></string-name>, <string-name><surname>He</surname> <given-names>X</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Deep spatio-temporal dependent convolutional LSTM network for traffic flow prediction</article-title>. <source>Sci Rep</source>. <year>2025</year>;<volume>15</volume>(<issue>1</issue>):<fpage>11743</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-025-95711-6</pub-id>; <pub-id pub-id-type="pmid">40189608</pub-id></mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Shin</surname> <given-names>J</given-names></string-name>, <string-name><surname>Miah</surname> <given-names>ASM</given-names></string-name>, <string-name><surname>Egawa</surname> <given-names>R</given-names></string-name>, <string-name><surname>Hassan</surname> <given-names>N</given-names></string-name>, <string-name><surname>Hirooka</surname> <given-names>K</given-names></string-name>, <string-name><surname>Tomioka</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Multimodal fall detection using spatial-temporal attention and Bi-LSTM-based feature fusion</article-title>. <source>Future Internet</source>. <year>2025</year>;<volume>17</volume>(<issue>4</issue>):<fpage>173</fpage>. doi:<pub-id pub-id-type="doi">10.3390/fi17040173</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Miah</surname> <given-names>ASM</given-names></string-name>, <string-name><surname>Al Mehedi Hasan</surname> <given-names>M</given-names></string-name>, <string-name><surname>Jang</surname> <given-names>SW</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>HS</given-names></string-name>, <string-name><surname>Shin</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Multi-stream general and graph-based deep neural networks for skeleton-based sign language recognition</article-title>. <source>Electronics</source>. <year>2023</year>;<volume>12</volume>(<issue>13</issue>):<fpage>2841</fpage>. doi:<pub-id pub-id-type="doi">10.3390/electronics12132841</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Miah</surname> <given-names>ASM</given-names></string-name>, <string-name><surname>Al Mehedi Hasan</surname> <given-names>M</given-names></string-name>, <string-name><surname>Shin</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Dynamic hand gesture recognition using multi-branch attention based graph and general deep learning model</article-title>. <source>IEEE Access</source>. <year>2023</year>;<volume>11</volume>:<fpage>4703</fpage>&#x2013;<lpage>16</lpage>. doi:<pub-id pub-id-type="doi">10.1109/access.2023.3235368</pub-id>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Ouyang</surname> <given-names>W</given-names></string-name></person-group>. <article-title>Disentangling and unifying graph convolutions for skeleton-based action recognition</article-title>. In: <conf-name>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2020 Jun 13&#x2013;19</year>; <publisher-loc>Seattle, WA, USA</publisher-loc>. <publisher-name>IEEE; 2020</publisher-name>. p. <fpage>140</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr42600.2020.00022</pub-id>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Song</surname> <given-names>YF</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Shan</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Stronger, faster and more explainable: a graph convolutional baseline for skeleton-based action recognition</article-title>. In: <conf-name>Proceedings of the 28th ACM International Conference on Multimedia</conf-name>; <year>2020</year>; <publisher-loc>Seattle, WA, USA</publisher-loc>. <publisher-name>ACM</publisher-name>. p. <fpage>1625</fpage>&#x2013;<lpage>33</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3394171.3413802</pub-id>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ye</surname> <given-names>F</given-names></string-name>, <string-name><surname>Pu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhong</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>D</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Dynamic GCN: context-enriched topology learning for skeleton-based action recognition</article-title>. In: <conf-name>Proceedings of the 28th ACM International Conference on Multimedia</conf-name>; <year>2020</year>; <publisher-loc>Seattle, WA, USA</publisher-loc>. <publisher-name>ACM</publisher-name>. p. <fpage>55</fpage>&#x2013;<lpage>63</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3394171.3413941</pub-id>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Yuan</surname> <given-names>C</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Deng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>W</given-names></string-name></person-group>. <article-title>Channel-wise topology refinement graph convolution for skeleton-based action recognition</article-title>. In: <conf-name>2021 IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name>; <year>2021 Oct 10&#x2013;17</year>; <publisher-loc>Montreal, QC, Canada</publisher-loc>. <publisher-name>IEEE; 2021</publisher-name>. p. <fpage>13339</fpage>&#x2013;<lpage>48</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICCV48922.2021.01311</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Cheng</surname> <given-names>K</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>He</surname> <given-names>X</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>W</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Skeleton-based action recognition with shift graph convolutional network</article-title>. In: <conf-name>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <publisher-loc>Seattle, WA, USA</publisher-loc>, 2020. pp. <fpage>180</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR42600.2020.00026</pub-id>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Plizzari</surname> <given-names>C</given-names></string-name>, <string-name><surname>Cannici</surname> <given-names>M</given-names></string-name>, <string-name><surname>Matteucci</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Skeleton-based action recognition via spatial and temporal transformer networks</article-title>. <source>Comput Vis Image Underst</source>. <year>2021</year>;<volume>208</volume>(<issue>3</issue>):<fpage>103219</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.cviu.2021.103219</pub-id>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Abduljalil</surname> <given-names>H</given-names></string-name>, <string-name><surname>Elhayek</surname> <given-names>A</given-names></string-name>, <string-name><surname>Marish Ali</surname> <given-names>A</given-names></string-name>, <string-name><surname>Alsolami</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Spatiotemporal graph autoencoder network for skeleton-based human action recognition</article-title>. <source>AI</source>. <year>2024</year>;<volume>5</volume>(<issue>3</issue>):<fpage>1695</fpage>&#x2013;<lpage>708</lpage>. doi:<pub-id pub-id-type="doi">10.3390/ai5030083</pub-id>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kumer Paul</surname> <given-names>S</given-names></string-name>, <string-name><surname>Ala Walid</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Rani Paul</surname> <given-names>R</given-names></string-name>, <string-name><surname>Uddin</surname> <given-names>MJ</given-names></string-name>, <string-name><surname>Rana</surname> <given-names>MS</given-names></string-name>, <string-name><surname>Kumar Devnath</surname> <given-names>M</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>An Adam based CNN and LSTM approach for sign language recognition in real time for deaf people</article-title>. <source>Bulletin EEI</source>. <year>2024</year>;<volume>13</volume>(<issue>1</issue>):<fpage>499</fpage>&#x2013;<lpage>509</lpage>. doi:<pub-id pub-id-type="doi">10.11591/eei.v13i1.6059</pub-id>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mulyono</surname> <given-names>IUW</given-names></string-name>, <string-name><surname>Kusumawati</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Susanto</surname> <given-names>A</given-names></string-name>, <string-name><surname>Sari</surname> <given-names>CA</given-names></string-name>, <string-name><surname>Islam</surname> <given-names>HMM</given-names></string-name>, <string-name><surname>Doheir</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Hiragana character classification using convolutional neural networks methods based on Adam, SGD, and RMSProps optimizer</article-title>. <source>Sci J Informatics</source>. <year>2024</year>;<volume>11</volume>(<issue>2</issue>):<fpage>467</fpage>&#x2013;<lpage>76</lpage>. doi:<pub-id pub-id-type="doi">10.15294/sji.v11i2.2313</pub-id>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Merrouchi</surname> <given-names>M</given-names></string-name>, <string-name><surname>Atifi</surname> <given-names>K</given-names></string-name>, <string-name><surname>Skittou</surname> <given-names>M</given-names></string-name>, <string-name><surname>Benyoussef</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Gadi</surname> <given-names>T</given-names></string-name></person-group>. <article-title>AutoLrOpt: an efficient optimizer using automatic setting of learning rate for deep neural networks</article-title>. <source>IEEE Access</source>. <year>2024</year>;<volume>12</volume>(<issue>8</issue>):<fpage>83154</fpage>&#x2013;<lpage>68</lpage>. doi:<pub-id pub-id-type="doi">10.1109/access.2024.3413043</pub-id>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ahda</surname> <given-names>FA</given-names></string-name>, <string-name><surname>Wibawa</surname> <given-names>AP</given-names></string-name>, <string-name><surname>Dwi Prasetya</surname> <given-names>D</given-names></string-name>, <string-name><surname>Arbian Sulistyo</surname> <given-names>D</given-names></string-name></person-group>. <article-title>Comparison of Adam optimization and RMS prop in minangkabau-Indonesian bidirectional translation with neural machine translation</article-title>. <source>JOIV Int J Inform Vis</source>. <year>2024</year>;<volume>8</volume>(<issue>1</issue>):<fpage>231</fpage>. doi:<pub-id pub-id-type="doi">10.62527/joiv.8.1.1818</pub-id>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Jiang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>M</given-names></string-name>, <string-name><surname>Gan</surname> <given-names>W</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>W</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>J</given-names></string-name></person-group>. <article-title>STM: spatiotemporal and motion encoding for action recognition</article-title>. In: <conf-name>2019 IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name>; <year> 2019 Oct 27&#x2013;Nov 2</year>; <publisher-loc>Seoul, Republic of Korea</publisher-loc>. <publisher-name>IEEE; 2019</publisher-name>. p. <fpage>2000</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICCV.2019.00209</pub-id>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Saoudi</surname> <given-names>EM</given-names></string-name>, <string-name><surname>Jaafari</surname> <given-names>J</given-names></string-name>, <string-name><surname>Andaloussi</surname> <given-names>SJ</given-names></string-name></person-group>. <article-title>Advancing human action recognition: a hybrid approach using attention-based LSTM and 3D CNN</article-title>. <source>Sci Afr</source>. <year>2023</year>;<volume>21</volume>(<issue>3</issue>):<fpage>e01796</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.sciaf.2023.e01796</pub-id>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Burton-Barr</surname> <given-names>J</given-names></string-name>, <string-name><surname>Fernando</surname> <given-names>B</given-names></string-name>, <string-name><surname>Rajan</surname> <given-names>D</given-names></string-name></person-group>. <article-title>Activation control of vision models for sustainable AI systems</article-title>. <source>IEEE Trans Artif Intell</source>. <year>2024</year>;<volume>5</volume>(<issue>7</issue>):<fpage>3470</fpage>&#x2013;<lpage>81</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TAI.2024.3372935</pub-id>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Kahatapitiya</surname> <given-names>K</given-names></string-name>, <string-name><surname>Arnab</surname> <given-names>A</given-names></string-name>, <string-name><surname>Nagrani</surname> <given-names>A</given-names></string-name>, <string-name><surname>Ryoo</surname> <given-names>MS</given-names></string-name></person-group>. <article-title>VicTR: video-conditioned text representations for activity recognition</article-title>. arXiv:2304.02560. <year>2023</year>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Ranasinghe</surname> <given-names>K</given-names></string-name>, <string-name><surname>Naseer</surname> <given-names>M</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>FS</given-names></string-name>, <string-name><surname>Ryoo</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Self-supervised video transformer</article-title>. arXiv:2112.01514. <year>2021</year>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Sarraf</surname> <given-names>S</given-names></string-name>, <string-name><surname>Kabia</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Optimal topology of vision transformer for real-time video action recognition in an end-to-end cloud solution</article-title>. <source>Mach Learn Knowl Extr</source>. <year>2023</year>;<volume>5</volume>(<issue>4</issue>):<fpage>1320</fpage>&#x2013;<lpage>39</lpage>. doi:<pub-id pub-id-type="doi">10.3390/make5040067</pub-id>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hussain</surname> <given-names>A</given-names></string-name>, <string-name><surname>Hussain</surname> <given-names>T</given-names></string-name>, <string-name><surname>Ullah</surname> <given-names>W</given-names></string-name>, <string-name><surname>Baik</surname> <given-names>SW</given-names></string-name></person-group>. <article-title>Vision transformer and deep sequence learning for human activity recognition in surveillance videos</article-title>. <source>Comput Intell Neurosci</source>. <year>2022</year>;<volume>2022</volume>(<issue>3</issue>):<fpage>3454167</fpage>. doi:<pub-id pub-id-type="doi">10.1155/2022/3454167</pub-id>; <pub-id pub-id-type="pmid">35419045</pub-id></mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hussain</surname> <given-names>A</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>SU</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>N</given-names></string-name>, <string-name><surname>Ullah</surname> <given-names>W</given-names></string-name>, <string-name><surname>Alkhayyat</surname> <given-names>A</given-names></string-name>, <string-name><surname>Alharbi</surname> <given-names>M</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Shots segmentation-based optimized dual-stream framework for robust human activity recognition in surveillance video</article-title>. <source>Alex Eng J</source>. <year>2024</year>;<volume>91</volume>(<issue>9</issue>):<fpage>632</fpage>&#x2013;<lpage>47</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.aej.2023.11.017</pub-id>.</mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Paul</surname> <given-names>SK</given-names></string-name>, <string-name><surname>Zisa</surname> <given-names>AA</given-names></string-name>, <string-name><surname>Ala Walid</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Zeem</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Paul</surname> <given-names>RR</given-names></string-name>, <string-name><surname>Haque</surname> <given-names>MM</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Human fall detection system using long-term recurrent convolutional networks for next-generation healthcare: a study of human motion recognition</article-title>. In: <conf-name>2023 14th International Conference on Computing Communication and Networking Technologies (ICCCNT)</conf-name>; <year> 2023 Jul 6&#x2013;8</year>; <publisher-loc>Delhi, India</publisher-loc>. <publisher-name>IEEE</publisher-name>; <volume>2023</volume>. p. <fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICCCNT56998.2023.10308247</pub-id>.</mixed-citation></ref>
</ref-list>
</back></article>