<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">67359</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2025.067359</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Deep Learning Models for Detecting Cheating in Online Exams</article-title>
<alt-title alt-title-type="left-running-head">Deep Learning Models for Detecting Cheating in Online Exams</alt-title>
<alt-title alt-title-type="right-running-head">Deep Learning Models for Detecting Cheating in Online Exams</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Essahraui</surname><given-names>Siham</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Lamaakal</surname><given-names>Ismail</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-3" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Maleh</surname><given-names>Yassine</given-names></name><xref ref-type="aff" rid="aff-2">2</xref><email>y.maleh@usms.ma</email></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Makkaoui</surname><given-names>Khalid El</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Bouami</surname><given-names>Mouncef Filali</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Ouahbi</surname><given-names>Ibrahim</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Almousa</surname><given-names>May</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western"><surname>AlQahtani</surname><given-names>Ali Abdullah S.</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-9" contrib-type="author">
<name name-style="western"><surname>El-Latif</surname><given-names>Ahmed A. Abd</given-names></name><xref ref-type="aff" rid="aff-5">5</xref><xref ref-type="aff" rid="aff-6">6</xref></contrib>
<aff id="aff-1"><label>1</label><institution>Multidisciplinary Faculty of Nador, Mohammed Premier University</institution>, <addr-line>Oujda, 60000</addr-line>, <country>Morocco</country></aff>
<aff id="aff-2"><label>2</label><institution>Laboratory LaSTI, ENSAK, Sultan Moulay Slimane University</institution>, <addr-line>Khouribga, 54000</addr-line>, <country>Morocco</country></aff>
<aff id="aff-3"><label>3</label><institution>Department of Information Technology, College of Computer and Information Sciences, Princess Nourah bint Abdulrahman University</institution>, <addr-line>P.O. Box 84428, Riyadh, 11671</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-4"><label>4</label><institution>College of Computer and Information Sciences, Prince Sultan University</institution>, <addr-line>Riyadh, 11586</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-5"><label>5</label><institution>EIAS Data Science Lab, College of Computer and Information Sciences, and Center of Excellence in Quantum and Intelligent Computing, Prince Sultan University</institution>, <addr-line>Riyadh, 11586</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-6"><label>6</label><institution>Department of Mathematics and Computer Science, Faculty of Science, Menoufia University</institution>, <addr-line>Shebin El-Koom, 32511</addr-line>, <country>Egypt</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Yassine Maleh. Email: <email>y.maleh@usms.ma</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2025</year></pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>23</day><month>09</month><year>2025</year></pub-date>
<volume>85</volume>
<issue>2</issue>
<fpage>3151</fpage>
<lpage>3183</lpage>
<history>
<date date-type="received">
<day>01</day>
<month>5</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>16</day>
<month>6</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 The Authors.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_67359.pdf"></self-uri>
<abstract>
<p>The rapid shift to online education has introduced significant challenges to maintaining academic integrity in remote assessments, as traditional proctoring methods fall short in preventing cheating. The increase in cheating during online exams highlights the need for efficient, adaptable detection models to uphold academic credibility. This paper presents a comprehensive analysis of various deep learning models for cheating detection in online proctoring systems, evaluating their accuracy, efficiency, and adaptability. We benchmark several advanced architectures, including EfficientNet, MobileNetV2, ResNet variants and more, using two specialized datasets (OEP and OP) tailored for online proctoring contexts. Our findings reveal that EfficientNetB1 and YOLOv5 achieve top performance on the OP dataset, with EfficientNetB1 attaining a peak accuracy of 94.59% and YOLOv5 reaching a mean average precision (mAP@0.5) of 98.3%. For the OEP dataset, ResNet50-CBAM, YOLOv5 and EfficientNetB0 stand out, with ResNet50-CBAM achieving an accuracy of 93.61% and EfficientNetB0 showing robust detection performance with balanced accuracy and computational efficiency. These results underscore the importance of selecting models that balance accuracy and efficiency, supporting scalable, effective cheating detection in online assessments.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Anti-cheating model</kwd>
<kwd>computer vision (CV)</kwd>
<kwd>deep learning (DL)</kwd>
<kwd>online exam proctoring</kwd>
<kwd>neural networks</kwd>
<kwd>facial recognition</kwd>
<kwd>biometric authentication</kwd>
<kwd>security of distance education</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>Princess Nourah bint Abdulrahman University</funding-source>
<award-id>PNURSP2025R752</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Examinations serve as a cornerstone of educational assessment, rigorously evaluating students&#x2019; comprehension, skills, and abilities across diverse subject areas. Traditionally administered within controlled, in-person settings, exams have been widely recognized as a credible measure of academic achievement, as well as a crucial determinant of students&#x2019; academic and professional pathways [<xref ref-type="bibr" rid="ref-1">1</xref>]. With the recent, unprecedented shift to online learning [<xref ref-type="bibr" rid="ref-2">2</xref>], educational institutions worldwide have been compelled to reimagine and adapt these evaluative frameworks. This transition, often to digital assessment platforms, has fundamentally transformed the nature of the assessment process, introducing critical challenges to maintaining fairness, equity, and academic honesty [<xref ref-type="bibr" rid="ref-3">3</xref>&#x2013;<xref ref-type="bibr" rid="ref-5">5</xref>].</p>
<p>The COVID-19 pandemic accelerated the widespread adoption of remote examinations, exposing inherent vulnerabilities within the online assessment environment [<xref ref-type="bibr" rid="ref-6">6</xref>,<xref ref-type="bibr" rid="ref-7">7</xref>]. Studies have highlighted a marked increase in cheating behaviors during online exams, underscoring the limitations of remote proctoring in preventing dishonest practices [<xref ref-type="bibr" rid="ref-8">8</xref>,<xref ref-type="bibr" rid="ref-9">9</xref>]. In the absence of physical oversight, combined with the high-stakes nature of assessments, students may be incentivized to exploit digital tools for assistance. The pervasive availability of internet resources, mobile devices, and screen-sharing technologies compounds this issue, posing significant risks to the credibility of exam outcomes and the integrity of academic assessments [<xref ref-type="bibr" rid="ref-10">10</xref>].</p>
<p>Several studies attribute the surge in online exam cheating to a range of factors, including environmental variables, institutional pressures, and the distinct challenges inherent to remote assessments [<xref ref-type="bibr" rid="ref-11">11</xref>]. As online education expands, ensuring the reliability of remote assessments has become a critical priority. However, existing solutions such as browser lockdown tools [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-13">13</xref>], identity verification processes [<xref ref-type="bibr" rid="ref-14">14</xref>], live proctoring [<xref ref-type="bibr" rid="ref-15">15</xref>], and 360 degree monitoring face notable limitations [<xref ref-type="bibr" rid="ref-16">16</xref>]. These methods often suffer from restricted dataset diversity, lack of real-time adaptability, and limited flexibility in addressing the wide array of cheating tactics that students may employ. Additionally, these approaches can be intrusive and challenging to implement consistently, particularly when external environmental factors are difficult to control, thus highlighting persistent gaps in the current strategies for online exam proctoring.</p>
<p>To address these challenges, the present study conducts an in-depth evaluation of various pre-trained deep learning models, leveraging two benchmark datasets specifically tailored for online proctoring contexts. By critically analyzing the performance, accuracy, and real-time adaptability of these models, this research aims to identify optimal approaches that achieve a balance between detection efficacy and computational efficiency. The findings presented herein offer valuable insights into the development of robust, scalable models that are capable of strengthening academic integrity within remote assessment environments.</p>
<p>The contributions of this paper are as follows:
<list list-type="simple">
<list-item><label>&#x2022;</label>
<p><bold>Comprehensive Comparison of Deep Learning Models for Cheating Detection:</bold> This study provides an in-depth comparative analysis of advanced deep learning models, including EfficientNet, MobileNetV2, and ResNet variants, specifically evaluating their effectiveness in detecting cheating behaviors in online exams.</p></list-item>
<list-item><label>&#x2022;</label>
<p><bold>Evaluation on Diverse Benchmark Datasets:</bold> By utilizing two distinct, specialized datasets (OEP and OP), the study ensures robust model assessment across varied online proctoring scenarios, contributing to a more generalized understanding of model performance in real-world online exam environments.</p></list-item>
<list-item><label>&#x2022;</label>
<p><bold>Guidelines for Model Selection Based on Accuracy and Efficiency:</bold> The paper offers valuable insights into model selection by balancing performance metrics like accuracy, F1 score, and computational efficiency, aiding institutions in choosing appropriate models for real-time proctoring systems based on their unique resource constraints.</p></list-item>
</list></p>
<p>The structure of this paper is organized as follows: <xref ref-type="sec" rid="s2">Section 2</xref> reviews related works, providing an overview of existing methodologies and their limitations. <xref ref-type="sec" rid="s3">Section 3</xref> details the datasets and classification methods used in this study. In <xref ref-type="sec" rid="s4">Section 4</xref>, we present evaluation metrics, experimental results, and an analysis of our findings. Then, we discuss the results in <xref ref-type="sec" rid="s5">Section 5</xref>, followed by the presentation of our work&#x2019;s limitations in <xref ref-type="sec" rid="s6">Section 6</xref>. Finally, <xref ref-type="sec" rid="s7">Section 7</xref> offers conclusions and suggests directions for future research.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Works</title>
<p>This section provides an overview of recent systems for cheating detection in online examinations, highlighting the application of machine learning, deep learning, and computer vision methodologies for visual analysis and behavioral surveillance.</p>
<p>Liu et al. [<xref ref-type="bibr" rid="ref-8">8</xref>] proposed a framwork named CHEESE for identifying academic dishonesty in online examinations by multiple instance learning (MIL) and spatio-temporal graph analysis. They introduced an innovative weakly supervised method that combines body posture, gaze, head movement, and face data, obtained from OpenFace 2.0 and 3D convolutional networks, to identify spatio-temporal abnormalities in video recordings. Evaluated on the Online Exam Proctoring (OEP) dataset of 65 aberrant and 69 normal videos, CHEESE attained an AUC of 87.58%, surpassing leading methodologies. The model exhibited robust performance on additional anomaly detection datasets, achieving 80.56% AUC on UCF-Crime and 90.79% on ShanghaiTech Dataset.</p>
<p>Ramzan et al. [<xref ref-type="bibr" rid="ref-17">17</xref>] evaluated the efficacy of pre-trained convolutional neural networks in identifying anomalous behaviors during online examinations, specifically cheating behaviors. Key video frames were recovered by motion-based approaches, and deep learning models, including YOLOv5, DenseNet121, and Inception-V3, were utilized to categorize cheating behaviors, such as head movements or the use of external devices. The YOLOv5 model attained superior performance, with a precision of 95.54%, a recall of 93.16%, and a mean average precision (mAP) of 95.40%, outpacing alternative methods in identifying cheating behaviors during online examinations.</p>
<p>Kaddoura and Gumaei [<xref ref-type="bibr" rid="ref-18">18</xref>] suggested a deep learning method for detecting cheating in online examinations using video frames and speech analysis. Their approach incorporates three essential modules: front camera detection, rear camera detection, and speech-based detection, utilizing CNN and Gaussian-based DFT for real-time cheating classification. The system was tested on the OEP dataset, which comprised data from 24 students, achieving an accuracy of 99.83% for front camera recognition and 98.78% for back camera detection.</p>
<p>Nurpeisova et al. [<xref ref-type="bibr" rid="ref-19">19</xref>] created the Proctor SU system, an automated online proctoring technology aimed at improving the integrity of examinations in the higher education sector in Kazakhstan. The system incorporates AI technologies like face detection, facial recognition, and behavioral analysis, employing models such as CNN, R-CNN, and YOLOv3 for real-time surveillance. The system, evaluated at Seifullin Kazakh Agro Technical University, analyzed 5000 photos with a detection rate of 97.12%.</p>
<p>Ozdamli et al. [<xref ref-type="bibr" rid="ref-20">20</xref>] created a facial recognition system to assess student emotions and identify cheating in remote learning settings. Their approach employed computer vision and deep learning technologies, such as CNN, to assess face expressions and actions during online examinations. The system, evaluated using datasets such as FER and FEI Face Database, attained an accuracy of 99.38% for student recognition and 66% for emotion detection. Furthermore, it monitored gaze direction and head motions to identify cheating, achieving a gaze tracking model accuracy of 96.95%.</p>
<p>Hu et al. [<xref ref-type="bibr" rid="ref-21">21</xref>] devoloped a multi-perspective adaptive cheating detection system for online examinations, employing three cameras (overhead, horizontal, and frontal views). The system integrates a gaze recognition model, detects cheating tools, and behavior analysis to enhance monitoring. The system employs datasets of 3000 photos for gaze direction and 7600 overhead and 8899 horizontal perspective images for detecting cheating tools, dynamically adjusting perspectives according to gaze direction. It attained 95% accuracy, providing an efficient and immediate method for mitigating cheating in digital examinations.</p>
<p>Dang et al. [<xref ref-type="bibr" rid="ref-22">22</xref>] developed an AI-driven auto-proctoring system intended for incorporation into MOOCs, emphasizing the identification of dishonest conduct during online assessments. The system integrates facial recognition, mobile device detection, and head posture estimation, employing deep learning models like RetinaFace and YOLOv10 for precise identification. Their system, evaluated on 4311 films from actual examinations, attained a remarkable accuracy of 95.66%, with an average reaction time of 0.517 s.</p>
<p>Potluri et al. [<xref ref-type="bibr" rid="ref-23">23</xref>] presented an automated AI-driven online proctoring system utilizing Attentive-Net to identify student misconduct during tests. The system has four modules: facial detection, multiple individual detection, facial spoofing, and head posture estimation. The authors employed the CIPL dataset alongside a bespoke dataset of over 200 movies exhibiting diverse behaviors to assess their model. Their technique attained an accuracy of 87% utilizing the Attentive-Net in conjunction with the Liveness Net and head position estimation.</p>
<p>Several studies have concentrated on cheating detection systems for online examinations, incorporating facial recognition, object detection, and head and gaze tracking methods. Authors in [<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>] both employed YOLOv3 for object detection in their systems. They introduced a system that combines facial recognition, tracking head movements, identifying objects, gaze and mouth tracking, background analysis, etc. Achieving an accuracy surpassing 80% in gaze tracking, mouth supervision, and object detection. Singh et al. [<xref ref-type="bibr" rid="ref-26">26</xref>] similarly utilized YOLO to detect dishonest activities, including face recognition, mouth tracking, and mobile device detection, to ensure academic integrity during virtual assessments.</p>
<p>Roy and Chanda [<xref ref-type="bibr" rid="ref-27">27</xref>] focused on developing a cost-effective, webcam-based eye-gaze estimation system for human-computer interaction (HCI) using a convolutional neural network (CNN) and Mediapipe for real-time facial identification. Parkhi et al. [<xref ref-type="bibr" rid="ref-28">28</xref>] developed a comprehensive examination monitoring system leveraging deep learning, incorporating face detection, face spoofing detection, object detection (YOLO), eye tracking, and head-pose estimation. Their system achieved a 90% success rate in tracking face and head movements. Gadkar et al. [<xref ref-type="bibr" rid="ref-29">29</xref>] presented an automated proctoring system that incorporates facial recognition, mouth aspect ratio analysis for speech detection, and audio detection to identify suspicious phrases or multiple individuals in the frame, using Haar cascade for face detection.</p>
<p>In terms of anomaly detection, Atabay and Hassanpour [<xref ref-type="bibr" rid="ref-30">30</xref>] proposed a semi-supervised method for online exam proctoring based on skeletal similarity. Their system segments exam videos by evaluating skeletal features and computing the similarity between training and test segments to identify abnormal behavior.</p>
<p>Other approaches focus on human pose estimation. Samir et al. [<xref ref-type="bibr" rid="ref-31">31</xref>] utilized TensorFlow PoseNet to monitor head positions and hand motions in real-time, achieving detection accuracies of 94% for head posture and 97% for hand motions.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Research Methodology</title>
<p>This section outlines the proposed benchmarking approaches. To clearly illustrate the implementation flow of our proposed cheating detection framework, we provide the following pseudocode (see Algorithm 1), which outlines the key steps involved&#x2014;from data preprocessing and annotation to model training and evaluation. The process begins by extracting frames from the input videos, followed by manual annotation of those frames based on observed behaviors. The data is then split into training, validation, and testing subsets. Finally, a selected deep learning model (e.g., EfficientNet, ResNet, or YOLOv5) is trained using the labeled data and evaluated on unseen samples to detect cheating behavior. This structured pipeline ensures a reproducible and transparent methodology.</p>
<fig id="fig-13">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-13.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Benchmark Datasets</title>
<p>In this study, we employ two benchmarking datasets, capturing diverse exam behaviors and cheating activities (see <xref ref-type="fig" rid="fig-1">Fig. 1</xref>). These datasets are crucial for training machine learning models to enhance online examination integrity.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Representative sample images from the OEP and OP datasets</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-1.tif"/>
</fig>
<sec id="s3_1_1">
<label>3.1.1</label>
<title>Automated Online Exam Proctoring (OEP)</title>
<p>The OEP dataset [<xref ref-type="bibr" rid="ref-32">32</xref>] from CVLab at Michigan State University consists of video and audio recordings of exam sessions, recorded using webcams from 24 subjects, to replicate diverse testing settings that include both standard exam conduct and particular cheating behaviors. Every recording is meticulously annotated to emphasize key behaviors, including eye movements, head position, auditory signals, and ambient alterations, which are vital for the automatic identification of probable cheating. The dataset is organized to support machine learning applications, including labeled instances for supervised learning and baseline data for unsupervised techniques. This dataset is crucial for the advancement of AI-driven proctoring solutions, improving the integrity of online tests.</p>
</sec>
<sec id="s3_1_2">
<label>3.1.2</label>
<title>Online Proctoring Dataset (OP)</title>
<p>The Proctor-Dataset [<xref ref-type="bibr" rid="ref-33">33</xref>] comprises 200 videos that capture a range of examinee behaviors during online examinations. The dataset contains videos depicting both proper conduct and various forms of cheating behavior. Some videos showcase a combination of misbehavior types, while others specifically capture candidates looking at other screens, using devices for assistance, or conversing with individuals. This annotated dataset, designed for deep learning frameworks, enables ProctorNet to reliably detect a range of suspicious behaviors.</p>
<p><xref ref-type="table" rid="table-1">Table 1</xref> presents the main features of online exam proctoring datasets, including data types, duration, annotated behaviors, and application domains.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Characteristics of online exam proctoring datasets</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="center">Name of dataset</th>
<th>Ref.</th>
<th align="center">Samples</th>
<th align="center">Data types</th>
<th align="center">Duration (avg.)</th>
<th align="center">Cheating behaviors annotated</th>
<th align="center">Resolution</th>
<th align="center">Domain of application</th>
</tr>
</thead>
<tbody>
<tr>
<td>Automated Online Exam Proctoring (OEP)</td>
<td>[<xref ref-type="bibr" rid="ref-32">32</xref>]</td>
<td>24 subjects</td>
<td>Video, audio, annotations</td>
<td>10&#x2013;30 min per session</td>
<td>Eye movements, head position, auditory signals, ambient alterations</td>
<td>720 p</td>
<td>Online exam integrity, behavioral analysis, audio-visual analysis, ambient context detection</td>
</tr>
<tr>
<td>Online Proctoring Dataset (OP)</td>
<td>[<xref ref-type="bibr" rid="ref-33">33</xref>]</td>
<td>200 videos</td>
<td>Video, annotations</td>
<td>5&#x2013;15 min per video</td>
<td>Multiple misbehavior types, e.g., screen-looking, device use, external conversation</td>
<td>480 p</td>
<td>Online exam integrity, behavioral analysis, device usage detection, interaction context monitoring</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Data Pre-Processing</title>
<p>The datasets employed in this study consist of video files that were systematically converted into individual frames to facilitate their application in model training. Each video was segmented into discrete frames, enabling a precise manual classification of each frame into one of two categories: cheating and not cheating. This classification process was essential in constructing a well-defined labeled dataset, serving as the basis for effective model training and thorough evaluation. Subsequently, the classified frames were allocated into distinct subsets designated for training and validation, ensuring a balanced dataset structure that supports rigorous performance assessment and reliable validation of the models developed in this research.</p>
<p>The classification of samples into &#x201C;Cheating&#x201D; and &#x201C;No Cheating&#x201D; categories was a critical step in preparing the datasets (OEP and OP) for supervised learning. Initially, each video was decomposed into individual image frames at a fixed sampling rate, ensuring a consistent temporal representation of student behavior. The classification process was conducted manually by domain experts familiar with online proctoring standards, using behavior-based annotation criteria.</p>
<p>For the OEP dataset, classification decisions were based on observable actions such as eye movement irregularities (e.g., repeated glancing away from the screen), unusual head positions, presence of additional auditory signals (e.g., background conversation or device sounds), and environmental changes (e.g., another person entering the frame). Frames depicting these behaviors were labeled as &#x201C;Cheating,&#x201D; while those showing the student facing the screen with stable gaze and a quiet, controlled environment were labeled as &#x201C;No Cheating&#x201D;.</p>
<p>In the case of the OP dataset, the criteria extended to include more diverse cheating scenarios such as use of mobile phones, visible interaction with unauthorized individuals, and frequent attention shifts toward off-screen devices. Each image frame was visually inspected and categorized accordingly.</p>
<p>To facilitate machine-readable labeling, we used the LabelImg annotation tool to create bounding boxes around suspicious regions (e.g., the student&#x2019;s face, hands, or surrounding objects) and assign class labels. Each labeled image was stored with its corresponding XML or YOLO-format file, indicating the category (&#x201C;Cheating&#x201D; or &#x201C;No Cheating&#x201D;) and the location of the observed behavior. In total, 11,581 images from the OEP dataset and 9609 from the OP dataset were annotated.</p>
<p>These annotated samples were then divided into training (70%), validation (20%), and test (10%) sets while preserving the class distribution. This well-structured and behavior-informed classification pipeline enabled the deep learning models to learn fine-grained distinctions between normal and suspicious activities, thereby enhancing the robustness and accuracy of the cheating detection systems.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Pre-Trained Deep Learning Algorithms</title>
<p><list list-type="bullet">
<list-item>
<p>Convolutional Neural Network [<xref ref-type="bibr" rid="ref-34">34</xref>] is a deep learning model particularly well-suited for image and video analysis due to its capability to automatically learn spatial hierarchies of features. CNNs consist of multiple layers, including convolutional, pooling, and fully connected layers, which together capture and combine various visual patterns.</p>
<p><bold>Key Parameters:</bold>
<list list-type="simple">
<list-item><label>&#x2013;</label>
<p><bold>Conv2D Layers:</bold> Filters &#x003D; {32, 64, 128, 256}; Kernel Size &#x003D; (3, 3); Stride &#x003D; (1, 1)</p></list-item>
<list-item><label>&#x2013;</label>
<p><bold>MaxPooling2D:</bold> Pool Size &#x003D; (2, 2); Stride &#x003D; (2, 2)</p></list-item>
<list-item><label>&#x2013;</label>
<p><bold>Dense Layers:</bold> Units &#x003D; 512</p></list-item>
<list-item><label>&#x2013;</label>
<p><bold>Dropout Rate:</bold> 0.1</p></list-item>
</list></p>
<p>The convolution operation at the core of CNNs involves a sliding kernel applied to the input matrix, computing element-wise multiplications and summing the results. Formally:
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mrow><mml:mtext>Conv</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:munderover><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mi>x</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:mi>x</mml:mi></mml:math></inline-formula> is the input matrix, <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mi>k</mml:mi></mml:math></inline-formula> is the kernel matrix, and <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:mi>m</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>n</mml:mi></mml:math></inline-formula> defines the kernel size. Following this, a Rectified Linear Unit (ReLU) activation introduces non-linearity:
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mrow><mml:mtext>ReLU</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<p>Max pooling then reduces spatial dimensions, enhancing computational efficiency by retaining only the highest value in each pooling window:
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mrow><mml:mtext>MaxPool</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
</list-item></list>
</p>
<p>
<list list-type="bullet">
<list-item>
<p>MobileNetV2 [<xref ref-type="bibr" rid="ref-35">35</xref>] is a lightweight convolutional neural network optimized for mobile and embedded vision applications, utilizing depthwise separable convolutions to significantly reduce the number of parameters while maintaining accuracy. The model applies depthwise convolutions with filters of sizes 32, 64, and 128, where each channel is convolved independently, followed by a pointwise convolution with a (1, 1) kernel size. It incorporates a dropout rate of 0.2 to prevent overfitting. The architecture consists of an input layer of shape (224, 224, 3), a preprocessing layer using MobileNetV2&#x2019;s preprocess_input function, and a frozen MobileNetV2 base producing an output of shape (7, 7, 1280). A GlobalAveragePooling2D layer is followed by a dropout layer (rate &#x003D; 0.2) and two dense layers&#x2014;one with 128 ReLU-activated units and an output layer with a single sigmoid-activated unit. Depthwise convolution applies filters to each input channel individually, while the subsequent pointwise convolution refines feature extraction using a weight matrix.</p></list-item>
<list-item>
<p>ResNet50 and ResNet101 [<xref ref-type="bibr" rid="ref-36">36</xref>] tackle the vanishing gradient problem in deep networks using skip connections, which allow gradients to bypass certain layers, ensuring stable training. ResNet50 and ResNet101 contain 50 and 101 layers, respectively, facilitating deep feature extraction while maintaining efficiency. Each residual block consists of convolutional layers with filters of sizes {64, 128, 256, 512} and a kernel size of (3, 3), followed by batch normalization and ReLU activation. The residual connection is mathematically expressed as:
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mi>F</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>x</mml:mi></mml:math></disp-formula>where <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:mi>F</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represents the residual mapping and <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mi>x</mml:mi></mml:math></inline-formula> is the input. Batch normalization stabilizes training by normalizing activations. The fine-tuned ResNet50 model comprises an input layer (224, 224, 3), a frozen ResNet50 base producing an output of (7, 7, 2048), a flattening operation, and a sigmoid-activated dense output layer. In contrast, ResNet101 employs additional dropout layers (rate &#x003D; 0.5), batch normalization, and ReLU-activated dense layers.</p></list-item>
<list-item>
<p>EfficientNetB0 and EfficientNetB1 [<xref ref-type="bibr" rid="ref-37">37</xref>] achieve optimal accuracy and efficiency through compound scaling, which uniformly increases the depth, width, and resolution. EfficientNetB0 and EfficientNetB1 differ slightly in scale, with B1 having a slightly larger network size and higher accuracy. Each model consists of convolutional layers with filters of sizes {32, 64, 128, 256}, and scaling coefficients <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mi>w</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:mi>d</mml:mi></mml:math></inline-formula>, and <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mi>r</mml:mi></mml:math></inline-formula> that control width, depth, and resolution, respectively. The activation function used is Swish. Compound scaling balances network dimensions as follows:
<disp-formula id="ueqn-5"><mml:math id="mml-ueqn-5" display="block"><mml:mrow><mml:mtext>Width</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mi>w</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mspace width="1em"></mml:mspace><mml:mrow><mml:mtext>Depth</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mi>d</mml:mi><mml:mo>,</mml:mo><mml:mspace width="1em"></mml:mspace><mml:mrow><mml:mtext>Resolution</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mi>r</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></disp-formula>where <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mi>w</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mi>d</mml:mi></mml:math></inline-formula>, and <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:mi>r</mml:mi></mml:math></inline-formula> are coefficients for network scaling. The fine-tuned EfficientNetB0 model consists of an input layer (224, 224, 3), a frozen EfficientNetB0 base with max pooling, dropout layers (rate &#x003D; 0.4 and 0.5), batch normalization, and dense layers with 256 ReLU-activated units, followed by a sigmoid-activated output layer.</p></list-item>
<list-item>
<p>ConvNeXt [<xref ref-type="bibr" rid="ref-38">38</xref>] is a modernized CNN that integrates design principles from transformer architectures, such as layer normalization and residual connections, to enhance performance on computer vision tasks. It consists of convolutional layers with filters of sizes {32, 64, 128, 256}, utilizing GELU (Gaussian Error Linear Unit) as the activation function and layer normalization for stabilization. Layer normalization ensures stable training by normalizing each layer&#x2019;s output, expressed mathematically as:
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03BC;</mml:mi></mml:mrow><mml:msqrt><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:mtext>&#x03B5;</mml:mtext></mml:msqrt></mml:mfrac></mml:math></disp-formula>where <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:mi>&#x03BC;</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula> are computed per layer. The fine-tuned ConvNeXt model includes an input layer of shape (image_size, image_size, image_channel), multiple Conv2D layers with ReLU activation, MaxPooling2D layers with a (2, 2) pool size, a flattening operation, a dense layer with 128 ReLU-activated units, and a final sigmoid-activated output layer.</p></list-item>
<list-item>
<p>SE-ResNet [<xref ref-type="bibr" rid="ref-39">39</xref>] enhances ResNet by incorporating Squeeze-and-Excitation (SE) blocks that recalibrate channel-wise feature responses, allowing the network to focus on the most informative channels. The SE block consists of an excitation layer composed of dense layers with ReLU and sigmoid activations, while a reduction ratio controls the bottleneck in the SE block. The squeeze operation computes global channel-wise statistics as follows:
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>H</mml:mi></mml:mrow></mml:munderover><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:math></disp-formula>where <italic>H</italic> and <italic>W</italic> denote the height and width of the feature maps. This is followed by an excitation operation that selectively enhances important features:
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:mrow><mml:mtext>ReLU</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:msub><mml:mi>W</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:msub><mml:mi>W</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:math></inline-formula> are learnable weight matrices, and <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:mi>&#x03C3;</mml:mi></mml:math></inline-formula> represents the sigmoid activation function. These operations enable SE-ResNet to improve feature representation and boost model performance.</p>
</list-item>
<list-item>
<p>ResNet50-CBAM enhances ResNet50 by incorporating a Convolutional Block Attention Module (CBAM) [<xref ref-type="bibr" rid="ref-40">40</xref>], which refines feature learning by focusing on both channel and spatial dimensions to highlight relevant information. The attention module consists of channel and spatial attention blocks, utilizing the Exponential Linear Unit (ELU) activation function. Channel attention emphasizes significant channels and is computed as:
<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:msub><mml:mi>M</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>FC</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:mi>&#x03C3;</mml:mi></mml:math></inline-formula> is the sigmoid function. Spatial attention, on the other hand, identifies important spatial locations using a convolutional operation:
<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:msub><mml:mi>M</mml:mi><mml:mi>s</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>f</mml:mi><mml:mrow><mml:mn>7</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>7</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:msup><mml:mi>f</mml:mi><mml:mrow><mml:mn>7</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>7</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> represents a convolution with a kernel size of <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:mn>7</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>7</mml:mn></mml:math></inline-formula>. These attention mechanisms enable ResNet50-CBAM to capture more informative features, improving classification performance.</p></list-item>
<list-item>
<p>Swin Transformer [<xref ref-type="bibr" rid="ref-41">41</xref>] is a vision transformer model tailored for computer vision tasks. It segments images into non-overlapping patches and applies self-attention within these patches, enhancing computational efficiency and scalability. The model hierarchically merges patches across layers, allowing for multi-scale feature representations. Key parameters include a patch size of (4, 4), a window size of 7, and attention heads ranging from 8 to 32. The self-attention mechanism within each window is computed as:
<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mrow><mml:mtext>Attention</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mtext>softmax</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>Q</mml:mi><mml:msup><mml:mi>K</mml:mi><mml:mi>T</mml:mi></mml:msup></mml:mrow><mml:msqrt><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:msqrt></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mi>V</mml:mi></mml:math></disp-formula>where <italic>Q</italic>, <italic>K</italic>, and <italic>V</italic> represent the query, key, and value matrices, respectively, and <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:math></inline-formula> denotes the dimension of the key vectors. A shifting window technique is employed to expand the receptive field while maintaining computational efficiency.</p></list-item>
<list-item>
<p>YOLOv5: YOLO [<xref ref-type="bibr" rid="ref-42">42</xref>] is a family of deep learning models optimized for real-time object detection, striking an effective balance between speed and accuracy. This model utilizes a 640 &#x00D7; 640 pixel image resolution, a batch size of 16, and is trained over 70 epochs to ensure optimal learning. It has been fine-tuned to detect cheating in a binary classification setting, distinguishing between cheating and not cheating.</p>
<p>For each detected instance, the YOLOv5s model outputs:
<list list-type="simple">
<list-item><label>&#x2013;</label><p><bold>Bounding box coordinates</bold> <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>w</mml:mi><mml:mo>,</mml:mo><mml:mi>h</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, where <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:mi>x</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mi>y</mml:mi></mml:math></inline-formula> represent the center of the bounding box, while <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mi>w</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:mi>h</mml:mi></mml:math></inline-formula> define its width and height.</p></list-item>
<list-item><label>&#x2013;</label><p><bold>Confidence score</bold> <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mtext>obj</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>, representing the probability of an object being present in the bounding box.</p></list-item>
<list-item><label>&#x2013;</label><p><bold>Class probabilities</bold> <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mtext>class</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>, indicating the likelihood of the detected object belonging to each possible class.</p>
<p>The final score for a class-specific detection is calculated as:
<disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mrow><mml:mtext>obj</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mrow><mml:mtext>class</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:math></disp-formula></p>
</list-item>
</list></p></list-item>
</list></p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Evaluation and Results</title>
<sec id="s4_1">
<label>4.1</label>
<title>Experimental Set-Up</title>
<p>This study examines the effectiveness of various deep learning algorithms&#x2014;including CNN, MobileNetV2, ResNet50, ResNet101, EfficientNetB0, EfficientNetB1, ConvNeXt, ResNet50_CBAM, SeResNet, Swin Transformer, and YOLOv5&#x2014;in identifying cheating behaviors during online examinations. Two datasets were utilized in the experiments. Each dataset was divided into 70% for training, 20% for validation, and 10% for testing to ensure optimal training and assessment.</p>
<p>The implementation was carried out using Python version 3.11 on a personal computer using an Intel Core i7-11800H CPU at 2.30 GHz and 16 GB of RAM. To accelerate the training process, the models utilized GPU resources from Google Colaboratory, notably an NVIDIA Tesla V100 GPU, which enabled efficient model optimization and faster convergence.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Evaluation Metrics</title>
<p>To assess the performance of a machine learning model in the context of cheating detection, a range of metrics are employed to gain a comprehensive understanding of the model&#x2019;s effectiveness, efficiency, and reliability.
<list list-type="bullet">
<list-item>
<p>Accuracy <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:mo stretchy="false">(</mml:mo><mml:mi>A</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>: The proportion of correctly identified instances, encompassing both cheats and non-cheats.
<disp-formula id="eqn-12"><label>(12)</label><mml:math id="mml-eqn-12" display="block"><mml:mi>A</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p></list-item>
<list-item>
<p>Recall <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:mo stretchy="false">(</mml:mo><mml:mi>R</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>: The fraction of actual cheats that the classifier correctly identifies.
<disp-formula id="eqn-13"><label>(13)</label><mml:math id="mml-eqn-13" display="block"><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p></list-item>
<list-item>
<p>Precision <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:mo stretchy="false">(</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>: The proportion of cheat predictions that are correct (true positives).
<disp-formula id="eqn-14"><label>(14)</label><mml:math id="mml-eqn-14" display="block"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p></list-item>
<list-item>
<p>F1-score <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:mo stretchy="false">(</mml:mo><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>: The harmonic mean of Recall and Precision, offering a balanced measure between them.
<disp-formula id="eqn-15"><label>(15)</label><mml:math id="mml-eqn-15" display="block"><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mfrac><mml:mrow><mml:mi>P</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>R</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p></list-item>
<list-item>
<p>Confusion Matrix <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:mo stretchy="false">(</mml:mo><mml:mi>C</mml:mi><mml:mi>M</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>: Summarizes classifier performance using four values: True Positive (TP), False Positive (FP), True Negative (TN), and False Negative (FN).</p></list-item>
<list-item>
<p>Mean Average Precision at IoU Threshold 0.5 <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:mo stretchy="false">(</mml:mo><mml:mi>m</mml:mi><mml:mi>A</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo>@</mml:mo></mml:mrow><mml:mn>0.5</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>: The average precision calculated at an Intersection over Union (IoU) threshold of 0.5, representing a looser overlap criterion.
<disp-formula id="eqn-16"><label>(16)</label><mml:math id="mml-eqn-16" display="block"><mml:mi>m</mml:mi><mml:mi>A</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo>@</mml:mo></mml:mrow><mml:mn>0.5</mml:mn><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:mi>A</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mspace width="thinmathspace"></mml:mspace><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mspace width="thinmathspace"></mml:mspace><mml:mrow><mml:mtext>IoU</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn></mml:mrow></mml:msub></mml:math></disp-formula></p></list-item>
<list-item>
<p>Mean Average Precision at IoU Range 0.5 to 0.95 <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:mo stretchy="false">(</mml:mo><mml:mi>m</mml:mi><mml:mi>A</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo>@</mml:mo></mml:mrow><mml:mn>0.5</mml:mn><mml:mo>:</mml:mo><mml:mn>0.95</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>: The mean average precision calculated over IoU thresholds ranging from 0.5 to 0.95 in 0.05 increments, providing a nuanced accuracy measure.
<disp-formula id="eqn-17"><label>(17)</label><mml:math id="mml-eqn-17" display="block"><mml:mi>m</mml:mi><mml:mi>A</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo>@</mml:mo></mml:mrow><mml:mn>0.5</mml:mn><mml:mo>:</mml:mo><mml:mn>0.95</mml:mn><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:mfrac><mml:mn>1</mml:mn><mml:mn>10</mml:mn></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn></mml:mrow><mml:mrow><mml:mn>0.95</mml:mn></mml:mrow></mml:munderover><mml:mi>A</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mspace width="thinmathspace"></mml:mspace><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mspace width="thinmathspace"></mml:mspace><mml:mrow><mml:mtext>IoU</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></disp-formula></p></list-item>
<list-item>
<p>Inference Time <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:mo stretchy="false">(</mml:mo><mml:mi>I</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>: The average time (in milliseconds) required for the model to process a single image during inference.</p></list-item>
<list-item>
<p>Training Time <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:mo stretchy="false">(</mml:mo><mml:mi>T</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>: The total time required to train the model on the dataset, from start to end.
<disp-formula id="eqn-18"><label>(18)</label><mml:math id="mml-eqn-18" display="block"><mml:mi>T</mml:mi><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mtext>end\_training</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mtext>start\_training</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:math></disp-formula></p></list-item>
<list-item>
<p>Prediction Time <inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:mo stretchy="false">(</mml:mo><mml:mi>P</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>: The time taken by the classifier to process the test dataset and categorize each instance as cheating or not.
<disp-formula id="eqn-19"><label>(19)</label><mml:math id="mml-eqn-19" display="block"><mml:mi>P</mml:mi><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mtext>end\_testing</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mtext>start\_testing</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:math></disp-formula></p></list-item>
</list></p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Result Analysis and Discussion</title>
<p>This section provides an in-depth assessment of the proposed deep learning models aimed at detecting cheating behavior in online examinations. It encompasses a detailed overview of the experimental setup, an in-depth discussion of the evaluation metrics utilized, and a comprehensive presentation of the results obtained.</p>
<sec id="s4_3_1">
<label>4.3.1</label>
<title>Performance on OEP Dataset</title>
<p><xref ref-type="fig" rid="fig-2">Figs. 2</xref> and <xref ref-type="fig" rid="fig-3">3</xref> display the accuracy and loss metrics for our proposed deep learning models utilizing the OEP dataset, demonstrating considerable variations among the models. The Swin Transformer and EfficientNetB1 models achieved the highest accuracy, indicating superior predictive performance, while models such as MobileNetV2 and ResNet50 demonstrated moderate accuracy but at a lower computational cost. The ConvNext model exhibited computational efficiency; nevertheless, its marginally reduced accuracy indicates possible trade-offs between model complexity and predictive capability.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Accuracy and loss for DL models using OEP dataset (Part 1)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-2a.tif"/>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-2b.tif"/>
</fig>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Accuracy and loss for DL models using OEP dataset (Part 2)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-3a.tif"/>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-3b.tif"/>
</fig>
<p>Models exhibiting superior accuracy, specifically EfficientNetB1 and the Swin Transformer, attained reduced loss function values, hence confirming their trustworthiness and efficacy in minimizing prediction mistakes. On the other hand, SE-ResNet and ResNet101 had high loss values, which meant that their predictive performance was not as stable as that of models such as ResNet50-CBAM and EfficientNetB0, which had a better balance between accuracy and loss. We identified the Swin Transformer and EfficientNetB1 as the leading models, which demonstrated both superior accuracy and little loss, underscoring their efficacy in trustworthy cheating detection during online examinations.</p>
<p><xref ref-type="fig" rid="fig-4">Fig. 4</xref> presents confusion matrices that provide a comparative investigation of several deep learning models for identifying cheating and non-cheating behaviors in online examinations. Each matrix displays the true positive (TP), true negative (TN), false positive (FP), and false negative (FN) rates, enabling an evaluation of the accuracy and recall of each model. EfficientNetB0 and ResNet50-CBAM exhibit the best accuracy, each attaining a 97% TN rate and an 88% TP rate, making them the most balanced models. EfficientNetB1 and ConvNext also perform well, with EfficientNetB1 attaining 97% accuracy for non-cheating and 84% for cheating, while ConvNext achieves 96% and 88%, respectively. MobileNetV2 and ResNet101 show moderate results, achieving around 95% accuracy in non-cheating detection but slightly lower cheating detection rates (83% and 82%).</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Confusion matrices for DL models using OEP dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-4.tif"/>
</fig>
<p>The Swin Transformer and SE-ResNet exhibit worse performance, with TP rates of 79% and 86%, respectively, and an increased incidence of FPs, suggesting less reliability in detecting instances of cheating. Overall, EfficientNetB0 and ResNet50-CBAM stand out as the optimal choices for precise and equitable cheating detection.</p>
<p><xref ref-type="table" rid="table-2">Table 2</xref> presents the performance of our proposed deep learning models on the OEP dataset, with significant metrics emphasized to identify the top-performing models. ResNet50 and ResNet50-CBAM stand out as leading contenders. ResNet50 attains the greatest test accuracy of 93.87%, an exceptional F1-Score of 91.75%, and an AUC of 93.01%. ResNet50-CBAM demonstrates a test accuracy of 93.70%, with a precision of 96.70%, an F1 score of 91.28%, and an AUC of 92.30%. These findings illustrate their equitable and resilient performance.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Performance metrics for DL models using the OEP dataset</title>
</caption>
<table>
<colgroup>
<col/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col/>
</colgroup>
<thead>
<tr>
<th colspan="11">OEP dataset</th>
</tr>
<tr>
<th>Algorithm</th>
<th align="center">Train acc. (%)</th>
<th align="center">Train loss</th>
<th align="center">Test acc. (%)</th>
<th align="center">Test loss</th>
<th align="center">Train time (s)</th>
<th align="center">Test time (s)</th>
<th align="center">Precision (%)</th>
<th align="center">Recall (%)</th>
<th align="center">F1 Score (%)</th>
<th>AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td>CNN</td>
<td><bold>99.83</bold></td>
<td>0.0084</td>
<td>92.53</td>
<td>1.6971</td>
<td>1507.49</td>
<td>15.39</td>
<td>91.59</td>
<td><bold>90.04</bold></td>
<td>90.81</td>
<td>92.15</td>
</tr>
<tr>
<td>MobileNetV2</td>
<td>92.28</td>
<td>0.1811</td>
<td>90.76</td>
<td><bold>0.2651</bold></td>
<td>1485.99</td>
<td>5.49</td>
<td>96.14</td>
<td>78.96</td>
<td>86.71</td>
<td>88.50</td>
</tr>
<tr>
<td>ResNet50</td>
<td>99.44</td>
<td>0.1032</td>
<td><bold>93.87</bold></td>
<td>6.3117</td>
<td><bold>1120.02</bold></td>
<td>4.01</td>
<td>94.27</td>
<td>89.37</td>
<td><bold>91.75</bold></td>
<td><bold>93.01</bold></td>
</tr>
<tr>
<td>ResNet101</td>
<td>95.63</td>
<td>0.3680</td>
<td>89.81</td>
<td>0.9478</td>
<td>2715.05</td>
<td>7.29</td>
<td>90.50</td>
<td>81.90</td>
<td>85.99</td>
<td>88.30</td>
</tr>
<tr>
<td>EfficientNetB0</td>
<td>96.89</td>
<td>0.0729</td>
<td>93.35</td>
<td>0.2289</td>
<td>1391.09</td>
<td>8.91</td>
<td>94.84</td>
<td>87.33</td>
<td>90.93</td>
<td>92.20</td>
</tr>
<tr>
<td>EfficientNetB1</td>
<td>96.44</td>
<td>0.0879</td>
<td>92.83</td>
<td>0.2579</td>
<td>1186.47</td>
<td>10.10</td>
<td>94.99</td>
<td>85.75</td>
<td>90.13</td>
<td>91.48</td>
</tr>
<tr>
<td>ConvNext</td>
<td>99.75</td>
<td><bold>0.0067</bold></td>
<td>92.75</td>
<td>0.8346</td>
<td>1307.27</td>
<td><bold>2.08</bold></td>
<td>93.03</td>
<td>87.56</td>
<td>90.21</td>
<td>91.75</td>
</tr>
<tr>
<td>SE-ResNet</td>
<td>91.83</td>
<td>0.2152</td>
<td>89.03</td>
<td>0.3299</td>
<td>1361.42</td>
<td>3.17</td>
<td>85.71</td>
<td>85.52</td>
<td>85.62</td>
<td>88.36</td>
</tr>
<tr>
<td>ResNet50-CBAM</td>
<td>99.63</td>
<td>0.0095</td>
<td>93.70</td>
<td>0.5114</td>
<td>1873.80</td>
<td>3.49</td>
<td><bold>96.70</bold></td>
<td>86.43</td>
<td>91.28</td>
<td>92.30</td>
</tr>
<tr>
<td>Swin Transformer</td>
<td>92.34</td>
<td>0.35</td>
<td>86.44</td>
<td>0.43</td>
<td>18578.34</td>
<td>10.25</td>
<td>86.55</td>
<td>76.63</td>
<td>81.29</td>
<td>84.60</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-2fn2" fn-type="other">
<p>Note: The bold text in the tables highlights the highest-performing results identified in our study.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>EfficientNetB0 has commendable performance, with a test accuracy of 93.35%, an AUC of 92.20%, and a robust F1 score of 90.93%. EfficientNetB1 demonstrates competitiveness, achieving a test accuracy of 92.83%, a precision of 94.99%, and an F1 score of 90.13%.</p>
<p>In terms of training efficiency, ResNet50 stands out with the lowest training duration of 1120.02 s while ConvNext attains the test accuracy of 92.75% with F1-score of 90.21%, achieving the quickest test time of 2.08 s, making it efficient for real-time applications. CNN demonstrates high training accuracy (99.83%) with a test accuracy 92.53%.</p>
<p>MobileNetV2 and ResNet101 demonstrate modest performance, achieving test accuracies of 90.76% and 89.81%, respectively; however, they underperform in F1 scores and precision. The Swin Transformer and SE-ResNet exhibit the poorest performance, with the Swin Transformer recording the lowest test accuracy at 86.44% and an AUC of 84.60%, indicating constrained efficacy in this application.</p>
<p>Overall, ResNet50 and ResNet50-CBAM deliver the best outcomes with high accuracy, precision, and F1 scores, while ConvNext excels in test efficiency, making these three models the most appropriate for effective cheating detection on the OEP dataset.</p>
</sec>
<sec id="s4_3_2">
<label>4.3.2</label>
<title>Performance on OP Dataset</title>
<p>The empirical study evaluates the performance of our proposed deep learning models for online cheating detection through an analysis of their accuracy and loss measures. Each model has distinct strengths and challenges in training dynamics, as seen in <xref ref-type="fig" rid="fig-5">Figs. 5</xref> and <xref ref-type="fig" rid="fig-6">6</xref> depicting accuracy and loss graphs.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Accuracy and loss for DL models using OP dataset (Part 1)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-5a.tif"/>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-5b.tif"/>
</fig>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Accuracy and loss for DL models using OP dataset (Part 2)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-6a.tif"/>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-6b.tif"/>
</fig>
<p>The EfficientNet variations (B0 and B1), SE-ResNet, and the Swin Transformer model demonstrate balanced convergence, attaining high accuracy with minimum loss, signifying strong generalization and stability. These models provide competitive performance while minimizing substantial overfitting, as evidenced by their consistent accuracy and minimal loss over the epochs. Conversely, models like CNN and ResNet101 exhibit more variance in accuracy and a slower convergence rate, potentially due to increased parameter complexity, which may hinder optimization.</p>
<p>ResNet50, ConvNext, and ResNet50-CBAM show distinct performance patterns when compared to the highest-performing models. ResNet50 and ConvNext provide consistent improvements in accuracy, although they display increased variability in loss during first training epochs, suggesting difficulties in achieving optimal convergence. ConvNext, featuring an innovative design influenced by CNN architecture, has somewhat superior accuracy stability compared to ResNet50; nonetheless, both models are slower to converge relative to EfficientNetB1 and SE-ResNet. To enhance accuracy compared to normal ResNet50 by optimizing feature emphasis through attention processes, the ResNet50-CBAM model is augmented with a Convolutional Block Attention Module (CBAM), yet it still falls short of the efficiency of top-performing models.</p>
<p>EfficientNetB1, SE-ResNet, and Swin Transformer are distinguished as the most proficient models, attaining elevated accuracy and little loss while exhibiting consistent learning curves.</p>
<p>The confusion matrices for the 10 deep learning models on the OP dataset offer a detailed assessment of their classification efficacy, shown in percentage format (see <xref ref-type="fig" rid="fig-7">Fig. 7</xref>).</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Confusion matrices for DL models using OP dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-7.tif"/>
</fig>
<p><list list-type="bullet">
<list-item>
<p>Leading Models: The highest-performing models are CNN (a), EfficientNetB1 (f), and ResNet50-CBAM (i), all demonstrating superior accuracy in identifying both categories. CNN achieves a TP rate of 97% and a TN rate of 91%, indicating balanced performance across both classes. EfficientNetB1 attains a high TP rate of 97%, albeit with a slightly lower TN rate of 84%. ResNet50-CBAM demonstrates dependable classification with a TP rate of 94% and a TN rate of 92%.</p></list-item>
<list-item>
<p>Robust Non-Cheating Detectors: Alternative models, including EfficientNetB0 (e), ResNet101 (d), and SE-ResNet (h), have shown efficacy in identifying Non-Cheating occurrences, with TP rates varying from 92% to 96%. Nonetheless, the Cheating detection rates exhibit variability; ResNet101 and EfficientNetB0 achieve commendable performance at about 92%, whilst SE-ResNet declines to 83% for this category, suggesting potential for improvement in achieving balanced detection.</p></list-item>
<list-item>
<p>Moderate Performers: MobileNetV2 (b) and ConvNext (g) exhibit moderate classification performance, each achieving a TP rate of 92%. Nevertheless, their accuracy in Cheating detection is marginally worse, with ConvNext at 90% and MobileNetV2 at 94%. These models exhibit consistent performance but fall short of the top performers in overall accuracy.</p></list-item>
<list-item>
<p>Lower Performing Model: The Swin Transformer (j) demonstrates the poorest overall performance, achieving a TP rate of just 79% in Non-Cheating detection, resulting in an elevated incidence of FP. Although it has a TN rate of 93%, demonstrating effective precision in identifying Cheating situations.</p></list-item>
</list></p>
<p>Analyzing the <xref ref-type="table" rid="table-3">Table 3</xref>, we observe distinct variances in performance across several models based on multiple metrics. ResNet50-CBAM attains the greatest train accuracy at 99.90%, while ConvNext follows closely with 99.79%. Test accuracy, which generally offers a better indication of model generalization, peaks with CNN at 97.19%, somewhat surpassing EfficientNetB1 and ResNet50. In terms of train loss, ResNet50-CBAM achieves the lowest score (0.0043), signifying excellent convergence during training, whereas EfficientNetB0 minimizes test loss (0.1436), showing strong performance on unseen data.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Performance metrics for DL models using the OP Dataset</title>
</caption>
<table>
<colgroup>
<col/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col/>
</colgroup>
<thead>
<tr>
<th colspan="11">OP dataset</th>
</tr>
<tr>
<th><bold>Algorithm</bold></th>
<th align="center"><bold>Train acc. (%)</bold></th>
<th align="center"><bold>Train loss</bold></th>
<th align="center"><bold>Test acc. (%)</bold></th>
<th align="center"><bold>Test loss</bold></th>
<th align="center"><bold>Train time (s)</bold></th>
<th align="center"><bold>Test time (s)</bold></th>
<th align="center"><bold>Precision (%)</bold></th>
<th align="center"><bold>Recall (%)</bold></th>
<th align="center"><bold>F1 score (%)</bold></th>
<th><bold>AUC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td>CNN</td>
<td>99.65</td>
<td>0.0131</td>
<td><bold>97.19</bold></td>
<td>0.4787</td>
<td>1302.89</td>
<td>13.48</td>
<td><bold>99.16</bold></td>
<td>95.36</td>
<td><bold>97.23</bold></td>
<td><bold>97.25</bold></td>
</tr>
<tr>
<td>MobileNetV2</td>
<td>96.46</td>
<td>0.0853</td>
<td>91.77</td>
<td>0.1903</td>
<td>1070.21</td>
<td>1.36</td>
<td>92.15</td>
<td>91.97</td>
<td>92.06</td>
<td>91.76</td>
</tr>
<tr>
<td>ResNet50</td>
<td>99.51</td>
<td>0.1454</td>
<td>93.75</td>
<td>4.9653</td>
<td><bold>849.21</bold></td>
<td>1.43</td>
<td>91.63</td>
<td><bold>96.79</bold></td>
<td>94.14</td>
<td>93.63</td>
</tr>
<tr>
<td>ResNet101</td>
<td>96.95</td>
<td>0.1090</td>
<td>92.08</td>
<td>0.6471</td>
<td>2165.60</td>
<td>1.93</td>
<td>92.54</td>
<td>92.17</td>
<td>92.35</td>
<td>92.08</td>
</tr>
<tr>
<td>EfficientNetB0</td>
<td>97.27</td>
<td>0.0632</td>
<td>93.13</td>
<td><bold>0.1436</bold></td>
<td>1162.60</td>
<td>1.50</td>
<td>95.18</td>
<td>91.37</td>
<td>93.24</td>
<td>93.19</td>
</tr>
<tr>
<td>EfficientNetB1</td>
<td>97.12</td>
<td>0.0724</td>
<td>94.48</td>
<td>0.1474</td>
<td>1064.67</td>
<td>1.39</td>
<td>94.95</td>
<td>94.38</td>
<td>94.66</td>
<td>94.48</td>
</tr>
<tr>
<td>ConvNext</td>
<td>99.79</td>
<td>0.0066</td>
<td>90.94</td>
<td>0.5454</td>
<td>1022.76</td>
<td><bold>1.35</bold></td>
<td>92.20</td>
<td>90.16</td>
<td>91.17</td>
<td>90.97</td>
</tr>
<tr>
<td>SE-ResNet</td>
<td>91.59</td>
<td>0.2030</td>
<td>89.48</td>
<td>0.2425</td>
<td>1204.85</td>
<td>1.36</td>
<td>95.84</td>
<td>83.33</td>
<td>89.15</td>
<td>89.72</td>
</tr>
<tr>
<td>ResNet50-CBAM</td>
<td><bold>99.90</bold></td>
<td><bold>0.0043</bold></td>
<td>93.23</td>
<td>0.3272</td>
<td>1535.16</td>
<td>1.38</td>
<td>93.91</td>
<td>92.97</td>
<td>93.44</td>
<td>93.24</td>
</tr>
<tr>
<td>Swin Transformer</td>
<td>94.89</td>
<td>0.31</td>
<td>89.69</td>
<td>0.42</td>
<td>15562.46</td>
<td>9.93</td>
<td>90.67</td>
<td>87.74</td>
<td>89.18</td>
<td>89.63</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In terms of speed, ResNet50 exhibits the least training duration (849.21 s), but ConvNext is the quickest during testing (1.35 s). CNN achieve the highest accuracy at 97.19%, while ResNet50 attains the maximum recall of 96.79%, which is vital for applications that prioritize the identification of true positives. CNN distinguishes itself with the greatest F1 score (97.23%), representing an optimal equilibrium of accuracy and recall, making it appropriate for balanced classification tasks. The AUC score is essential for assessing the discriminative capability of a model, is highest (97.25%) for CNN.</p>
<p>In conclusion, CNN and ResNet50 emerge as top models based on a combination of high test accuracy, balanced F1 score, and competitive AUC. CNN demonstrates superior test accuracy, elevated F1 Score, and little test loss.</p>
</sec>
<sec id="s4_3_3">
<label>4.3.3</label>
<title>Performance on Yolov5 Model</title>
<p><xref ref-type="fig" rid="fig-8">Fig. 8</xref> compares the training accuracy and loss of the YOLOv5 model on two datasets. The model performs strongly on the OEP dataset, showing efficient convergence with all training and validation loss curves decreasing steadily and nearing zero, indicating effective learning and minimal overfitting. Precision and recall quickly stabilize around 0.95, demonstrating high accuracy in object detection and identification. Additionally, the mAP at IoU 0.5 reaches approximately 0.98, while mAP at IoU 0.5:0.95 stabilizes around 0.9, highlighting robust detection performance even under stricter conditions. Overall, the model exhibits high accuracy, efficient loss reduction, and strong generalization on the OEP dataset.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Training accuracy and loss using the YOLOv5 method</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-8.tif"/>
</fig>
<p>In the OP dataset, the YOLOv5 model shows effective training and validation performance with low and consistently decreasing loss values. All loss curves start at higher values but quickly converge towards zero, indicating efficient learning and minimizing overfitting. Validation losses similarly converge near zero, confirming strong generalization to the validation set. The precision and recall metrics stabilize around 0.95, demonstrating high accuracy in detecting and identifying objects. The mAP at IoU 0.5 reaches about 0.95, while mAP at IoU 0.5:0.95 stabilizes around 0.9, reflecting robust performance under varying IoU thresholds. Overall, these results suggest that the OP dataset enables efficient learning and convergence, with the model achieving high accuracy and low loss.</p>
<p>The YOLOv5 model performs excellently on both the OEP and OP datasets, with the best performance observed on the OEP dataset. It achieves high accuracy, efficient loss reduction, and strong generalization, with robust precision, recall, and mAP scores across both datasets.</p>
<p>The confusion matrices for the OEP and OP datasets (see <xref ref-type="fig" rid="fig-9">Fig. 9</xref>) demonstrate the efficacy of the model in categorizing no_cheating and cheating, emphasizing both strengths and weaknesses within each dataset.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>Confusion matrices using the YOLOv5 method</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-9.tif"/>
</fig>
<p>In the OEP dataset (see <xref ref-type="fig" rid="fig-9">Fig. 9a</xref>), the model demonstrates effective performance in distinguishing between no_cheating and cheating occurrences, achieving 95% accuracy for no_cheating and 93% for cheating. The model faces difficulties with background elements, misclassifying 45% of these cases as either no_cheating or cheating leading to a background accuracy of 55%. This indicates that background characteristics may exhibit similarities with other categories, resulting in misclassifications.</p>
<p>In the OP dataset (see <xref ref-type="fig" rid="fig-9">Fig. 9b</xref>), the model attains an accuracy of 95% in both the no_cheating and cheating categories. However, it encounters greater difficulty with background components, with a 54% misclassification rate and a 46% effective background accuracy. The somewhat elevated confusion rate relative to the OEP dataset suggests that background components in the OP dataset may be less distinguishable from no_cheating and cheating occurrences.</p>
<p>Both datasets provide a high accuracy of 95% in the no_cheating and cheating categories. The OEP dataset has a marginal superiority in managing background components, achieving an effective accuracy of 55% compared to 46% in the OP dataset, signifying more significant separations between background and labeled occurrences in the OEP dataset.</p>
<p><xref ref-type="fig" rid="fig-10">Fig. 10</xref> presents the precision-recall curves demonstrating the performance of the YOLOv5 model on the OEP and OP datasets for identifying the no_cheating and cheating classes. These graphs provide insights into the balance between accuracy and recall at various choice thresholds, Underscoring the effectiveness of the model in differentiating between these categories.</p>
<fig id="fig-10">
<label>Figure 10</label>
<caption>
<title>Precision-recall curve using YOLOv5</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-10.tif"/>
</fig>
<p>In the OEP dataset (see <xref ref-type="fig" rid="fig-10">Fig. 10a</xref>), the model attains elevated precision and recall metrics, with a precision-recall area of 0.985 for &#x201C;no_cheating&#x201D; and 0.978 for &#x201C;cheating.&#x201D; The mAP at a threshold of 0.5 across all classes is 0.981, signifying a robust equilibrium between accuracy and recall. The curves for each class are positioned around the upper right corner of the graph, indicating that the model excels in both recognizing TP and reducing FP for this dataset.</p>
<p>In the OP dataset (see <xref ref-type="fig" rid="fig-10">Fig. 10b</xref>), the model demonstrates robust performance, with a precision-recall area of 0.977 for &#x201C;no_cheating&#x201D; and a slightly higher 0.988 for &#x201C;cheating.&#x201D; The mAP@0.5 for all classes in the OP dataset is 0.983, just exceeding that of the OEP dataset. Analogous to the OEP results, the curves for each class remain around the optimal top right corner, signifying that the model adeptly balances accuracy and recall.</p>
<p>Both datasets have elevated accuracy and recall, with the OP dataset exhibiting a slightly superior mAP@0.5 (0.983) relative to the OEP dataset (0.981). This indicates that the YOLOv5 model is adept at identifying cheating and non-cheating events in both datasets, with marginally superior overall performance on the OP dataset.</p>
<p><xref ref-type="table" rid="table-4">Table 4</xref> compares the performance of YOLOv5 on the OEP and OP datasets across many metrics throughout the training and testing stages.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>YOLOv5 metrics for training and testing phases on OEP and OP datasets</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Metric</th>
<th colspan="2">OEP dataset</th>
<th colspan="2">OP dataset</th>
</tr>
<tr>
<th></th>
<th>Training</th>
<th>Testing</th>
<th>Training</th>
<th>Testing</th>
</tr>
</thead>
<tbody>
<tr>
<td>Precision (%)</td>
<td><bold>95.7</bold></td>
<td><bold>94.9</bold></td>
<td>94.1</td>
<td>93.8</td>
</tr>
<tr>
<td>Recall (%)</td>
<td>95.9</td>
<td>94.2</td>
<td><bold>96.0</bold></td>
<td><bold>96.2</bold></td>
</tr>
<tr>
<td>F1 Score (%)</td>
<td><bold>95.8</bold></td>
<td>94.5</td>
<td>95.0</td>
<td><bold>94.9</bold></td>
</tr>
<tr>
<td>mAP@0.5 (%)</td>
<td><bold>98.7</bold></td>
<td>98.1</td>
<td>98.1</td>
<td><bold>98.3</bold></td>
</tr>
<tr>
<td>mAP@0.5:0.95 (%)</td>
<td>97.5</td>
<td>97.0</td>
<td>97.1</td>
<td><bold>97.5</bold></td>
</tr>
<tr>
<td>Inference Time (ms)</td>
<td>7.3</td>
<td><bold>1.7</bold></td>
<td><bold>7</bold></td>
<td>1.8</td>
</tr>
<tr>
<td>Total Time (seconds)</td>
<td>4928.860</td>
<td>213.73</td>
<td><bold>3768.39</bold></td>
<td><bold>163.6</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The YOLOv5 model trained on the OEP dataset outperforms the one trained on the OP dataset in terms of accuracy, achieving 95.7% in training and 94.9% in testing, compared to 94.1% and 93.8% for the OP dataset model. However, the OP dataset model excels in recall, with rates of 96.0% in training and 96.2% in testing, slightly higher than the OEP model achieved 95.9% and 94.2%. This highlights the OP dataset&#x2019;s strength in identifying all relevant occurrences. Additionally, the OEP model shows a stronger F1 score in training at 95.8%, compared to 95.0% for the OP model, reflecting a more balanced performance between precision and recall.</p>
<p>In terms of mAP, YOLOv5 trained on the OEP dataset excels in mAP@0.5 during both training (98.7% vs. 98.1%) and testing (98.1% vs. 98.3%). However, the YOLOv5 model with the OP dataset outperforms in mAP@0.5:0.95 during testing, achieving 97.5% compared to the OEP model&#x2019;s 97.0%. Additionally, the OEP model demonstrates a slightly faster inference time during testing (1.7 vs. 1.8 ms), while the OP model significantly reduces the overall training time, completing in 3768.39 s compared to the OEP model&#x2019;s 4928.86 s, highlighting its greater training efficiency.</p>
<p>Overall, the OEP dataset displays greater accuracy and F1 scores, but the OP dataset excels in recall and efficiency, especially with inference and training length.</p>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Discussion</title>
<p>The evaluation of deep learning models on the OEP and OP datasets uncovers substantial insights into their efficacy for detecting online test cheating, highlighting the trade-offs among accuracy, computational efficiency, and resilience. EfficientNetB0 and ResNet50-CBAM continuously rank as superior models on the OEP dataset, exhibiting elevated test accuracy alongside balanced precision, F1 scores, and AUC, making them highly reliable choices for accurately detecting cheating behaviors. The ConvNext model, while not achieving the maximum accuracy, exhibits the quickest inference time, indicating its appropriateness for real-time applications. In the OP dataset, EfficientNetB1 excels in accuracy and F1 scores, demonstrating a balanced and effective performance in precision and recall measures, but with increased processing requirements during training and efficient test performance ensures reliable detection with minimal delay, aligning well with the demands of real-time applications (see <xref ref-type="fig" rid="fig-11">Figs. 11</xref> and <xref ref-type="fig" rid="fig-12">12</xref>).</p>
<fig id="fig-11">
<label>Figure 11</label>
<caption>
<title>Test accuracy and test time comparison for DL algorithms using OEP dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-11.tif"/>
</fig><fig id="fig-12">
<label>Figure 12</label>
<caption>
<title>Test accuracy and test time comparison for DL algorithms using OP dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_67359-fig-12.tif"/>
</fig>
<p>Concerning YOLOv5 on both datasets highlights the disparities in dataset architecture and its impact on model convergence. The OEP dataset seems to promote more efficient model convergence, probably owing to balanced feature patterns and class distributions, leading to consistently reduced test losses. In contrast, models trained on the OP dataset exhibit elevated loss values and variability, suggesting potential issues such as class imbalance or noisy features. Despite these limitations, YOLOv5 attains elevated precision-recall regions and mAP scores across both datasets, confirming its formidable detection capability.</p>
<p>The findings indicate that EfficientNetB0, EfficientNetB1, and ResNet50-CBAM are superior models for achieving a balance between accuracy, computational efficiency, and generalization, whereas YOLOv5 has significant potential for high-stakes cheating detection applications. The findings underscore the significance of choosing the suitable model and dataset, taking into account model convergence, efficiency, and class distribution to get dependable performance in identifying cheating behaviors.</p>
<p>To ensure practical applicability, the models evaluated in this study were specifically chosen and fine-tuned to handle a wide range of real-world cheating tactics, including behaviors involving second devices, behind-the-scenes assistance, and deceptive strategies like imitation. In the OP dataset, for instance, cheating behaviors include the examinee interacting with hidden mobile phones, receiving whispered answers, or mimicking natural behavior to avoid detection&#x2014;such as looking forward while covertly using peripheral vision to check another screen. Our use of object detection models like YOLOv5 enables direct localization of external devices (e.g., phones or tablets) and subtle cues such as head turns or partial hand gestures, making the system responsive to behind-the-scenes activity.</p>
<p>Furthermore, to simulate practical diversity, the datasets inherently contained variations in environmental conditions&#x2014;such as differing lighting setups (natural light, artificial overhead, or low-light rooms), background complexity, and camera angles (e.g., slightly above, straight-on, or tilted webcam views). Models like EfficientNetB1 and ResNet50-CBAM demonstrated strong resilience to these conditions due to their ability to learn fine-grained spatial features and adaptively focus on relevant regions through attention mechanisms. CBAM, in particular, improved robustness by dynamically emphasizing spatial and channel-level features, helping detect unusual posture shifts or light-induced image distortions.</p>
<p>The models also addressed imitation behavior&#x2014;where users attempt to mimic &#x201C;normal&#x201D; behavior&#x2014;through temporal consistency and anomaly detection across frames. For example, even if a student momentarily behaves correctly, sudden or inconsistent gestures, frequent gaze switching, or atypical face orientations across time frames are flagged as suspicious. Additionally, ResNet and ConvNeXt models&#x2019; deep feature extraction capabilities helped distinguish genuine from feigned attentiveness.</p>
<p>Together, this model ensemble supports real-time, context-aware classification, accounting for varied cheating strategies and ensuring adaptability across diverse testing environments. These robustness features are essential for deploying reliable proctoring systems in practical academic settings.</p>
<p><xref ref-type="table" rid="table-5">Table 5</xref> presents a detailed comparative analysis between our experimental outcomes and prior research on cheating detection in online examination contexts. The results demonstrate that our evaluated models generally exceed or closely align with state-of-the-art findings reported in existing literature. Specifically, our CNN implementation achieved an accuracy of 97.19% on the OP dataset, significantly outperforming the 93% reported in [<xref ref-type="bibr" rid="ref-19">19</xref>]. Similarly, our ResNet50 model achieved 93.87% on the OEP dataset and 93.75% on OP, surpassing the previously documented 87% in [<xref ref-type="bibr" rid="ref-23">23</xref>].</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Performance comparison of classifiers across two datasets</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Reference</th>
<th>Classifier</th>
<th>Year</th>
<th>Reported accuracy (%)</th>
<th colspan="2">Our Accuracy (%)</th>
</tr>
<tr>
<th/>
<th/>
<th/>
<th/>
<th>OEP dataset</th>
<th>OP dataset</th>
</tr>
</thead>
<tbody>
<tr>
<td>[<xref ref-type="bibr" rid="ref-19">19</xref>]</td>
<td>CNN</td>
<td>2023</td>
<td>93.00</td>
<td>92.53</td>
<td><bold>97.19</bold></td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-23">23</xref>]</td>
<td>ResNet50</td>
<td>2023</td>
<td>87.00</td>
<td><bold>93.87</bold></td>
<td><bold>93.75</bold></td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-43">43</xref>]</td>
<td>ConvNeXt</td>
<td>2023</td>
<td>88.82</td>
<td><bold>92.75</bold></td>
<td><bold>90.9</bold></td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-44">44</xref>]</td>
<td>Swin Transformer</td>
<td>2024</td>
<td><bold>90.42</bold></td>
<td>86.44</td>
<td>89.69</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-17">17</xref>]</td>
<td>YOLOv5</td>
<td>2024</td>
<td><bold>95.70</bold></td>
<td>94.50</td>
<td>94.90</td>
</tr>
<tr>
<td colspan="2"><bold>Additional Classifiers</bold></td>
<td></td>
<td></td>
<td></td>
<td></td>
</tr>
<tr>
<td colspan="2">SE-ResNet</td>
<td>&#x2013;</td>
<td>&#x2013;</td>
<td><bold>89.03</bold></td>
<td><bold>89.48</bold></td>
</tr>
<tr>
<td colspan="2">ResNet101</td>
<td>&#x2013;</td>
<td>&#x2013;</td>
<td><bold>89.81</bold></td>
<td><bold>92.08</bold></td>
</tr>
<tr>
<td colspan="2">MobileNetV2</td>
<td>&#x2013;</td>
<td>&#x2013;</td>
<td><bold>90.76</bold></td>
<td><bold>91.77</bold></td>
</tr>
<tr>
<td colspan="2">EfficientNetB1</td>
<td>&#x2013;</td>
<td>&#x2013;</td>
<td><bold>92.83</bold></td>
<td><bold>94.48</bold></td>
</tr>
<tr>
<td colspan="2">EfficientNetB0</td>
<td>&#x2013;</td>
<td>&#x2013;</td>
<td><bold>93.35</bold></td>
<td><bold>93.13</bold></td>
</tr>
<tr>
<td colspan="2">ResNet50-CBAM</td>
<td>&#x2013;</td>
<td>&#x2013;</td>
<td><bold>93.70</bold></td>
<td><bold>93.23</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>While ConvNext and Swin Transformer have been investigated in earlier works, our study examines them using more comprehensive datasets (OEP and OP), attaining 92.75% and 90.94% accuracy on OP for ConvNext and Swin Transformer, respectively. These results represent an improvement over earlier findings (88.82% for ConvNext [<xref ref-type="bibr" rid="ref-43">43</xref>] and 90.42% for Swin Transformer [<xref ref-type="bibr" rid="ref-44">44</xref>]). Furthermore, YOLOv5 continued to perform among the top models, consistent with prior benchmarks (95.70%), achieving 94.50% and 94.90% accuracy on the OEP and OP datasets, respectively [<xref ref-type="bibr" rid="ref-17">17</xref>].</p>
<p>In addition to revisiting previously studied classifiers, our work introduces and thoroughly assesses several advanced deep learning models not previously explored in the literature, including EfficientNetB0/B1, MobileNetV2, ResNet50-CBAM, and SE-ResNet. These models exhibited strong and dependable performance across both datasets, with EfficientNetB0 and EfficientNetB1 consistently surpassing 93% accuracy. Notably, ResNet50-CBAM reached 93.70% accuracy on OEP and 93.23% on OP, underscoring its effectiveness in balancing accuracy with architectural complexity.</p>
<p>Recent studies such as Automated Smart Artificial Intelligence-Based Proctoring System Using Deep Learning and Smart Artificial Intelligence-Based Online Proctoring System [<xref ref-type="bibr" rid="ref-45">45</xref>] have made significant strides in addressing cheating detection during online exams by developing a multi-modal system with pre-defined behavioral rules or heuristic methods to detect anomalies such as gaze deviation, head pose changes, or presence of multiple faces. While these approaches are effective in controlled environments, they often rely on limited behavior patterns, fixed camera conditions, and constrained datasets, which can reduce adaptability to real-world scenarios with high variability.</p>
<p>In contrast, our approach advances beyond these frameworks in several key ways. First, we introduce a diverse ensemble of advanced models&#x2014;including EfficientNetB0/B1, ResNet50-CBAM, YOLOv5, ConvNeXt, and Swin Transformer&#x2014;each offering unique architectural strengths such as compound scaling (EfficientNet) or attention-based refinement (CBAM). This allows our models to capture subtle cheating behaviors such as screen-looking, device use, and background interference with greater sensitivity and generalization.</p>
<p>Second, while previous smart AI-based systems typically focus on single-camera views and assume stable lighting or clean backgrounds, we benchmark our models using two robust, real-world datasets (OEP and OP) that include diverse lighting conditions, multiple angles, background noise, and imitation behaviors, making our evaluation more representative of actual online exam environments.</p>
<p>Third, our approach is deeply comparative and practical&#x2014;we not only report accuracy and F1-score but also assess each model&#x2019;s computational cost and inference time, which is crucial for deployability in real-time exam monitoring systems. This comprehensive evaluation makes our framework not only more scalable and accurate but also more adaptable to different institutional needs compared to existing smart AI-based systems.</p>
<p>Overall, the integration of these diverse and contemporary architectures into our study advances the field of deep learning-based cheating detection, providing compelling evidence for the adaptability and scalability of modern models in supporting academic integrity. Our findings underscore the value of examining a broad spectrum of architectures and datasets to identify models that best meet the operational demands of real-world proctoring environments.</p>
<p>The novelty of our proposed approach lies not only in the individual deep learning models used but also in the way we comprehensively benchmark and adapt them to the context of cheating detection under realistic, diverse, and challenging proctoring conditions. While some models like CNN and ResNet have been used in prior studies, our contribution advances the state of the art by introducing and evaluating more recent and powerful architectures&#x2014;including EfficientNetB0/B1, ResNet50-CBAM, ConvNeXt, and YOLOv5&#x2014;in a unified framework across two realistic datasets (OEP and OP). EfficientNet models introduce compound scaling, which allows for efficient depth, width, and resolution balancing, improving performance under constrained environments such as remote proctoring. ResNet50-CBAM integrates attention mechanisms at both the channel and spatial level, making it especially effective for detecting subtle and deceptive behaviors that might escape simpler models. YOLOv5, known for real-time object detection, is fine-tuned for binary classification of cheating behavior, providing both speed and high precision in detecting visual cues such as secondary devices or off-screen glances.</p>
<p>Furthermore, our work is novel in how it tests model robustness against variations such as lighting conditions, webcam angles, imitation behavior, and environmental distractions. Most prior works do not systematically evaluate model resilience in such diverse cheating scenarios. The comparative evaluation across architectures also provides practical insights into balancing accuracy and computational efficiency, making this work a step forward toward scalable, real-world deployment of AI-based proctoring systems.</p>
</sec>
<sec id="s6">
<label>6</label>
<title>Limitations of Our Work</title>
<p>While our study offers a comprehensive evaluation of deep learning models for cheating detection in online examinations, several limitations must be acknowledged. We primarily focused on CNN-based architectures such as EfficientNet, ResNet variants, and YOLOv5, without fully exploring advanced transformer-based models like Vision Transformers (ViT), DeiT, or hybrid CNN-transformer approaches, which have shown promising results in recent computer vision research. Additionally, our evaluation was limited to two datasets (OEP and OP), which, despite being relevant to online proctoring, may not fully capture the diversity of real-world examination settings in terms of behavior, demographics, and environmental context. Another limitation lies in the unimodal nature of our input data, as we relied solely on visual information from video frames. Incorporating multimodal inputs such as audio signals, gaze tracking, or keystroke dynamics could enhance model robustness and improve detection accuracy in complex scenarios.</p>
</sec>
<sec id="s7">
<label>7</label>
<title>Conclusion and Future Work</title>
<p>This paper presents a comprehensive analysis of deep learning models for detecting cheating in online exams, identifying EfficientNetB0, EfficientNetB1, ResNet50-CBAM, and YOLOv5 as top-performing architectures across the OEP and OP datasets. EfficientNet models and ResNet50-CBAM demonstrated balanced accuracy, efficiency, and strong precision-recall performance, making them suitable for real-time proctoring applications, while YOLOv5 excelled in precision and mAP, showcasing potential for high-stakes, real-time detection. Future work should aim to enhance model robustness through hybrid and ensemble approaches, expand datasets to capture diverse cheating behaviors, and develop lightweight, privacy-preserving, multi-modal frameworks that integrate audio, gaze, and other behavioral cues. Addressing ethical concerns, refining adaptive feedback mechanisms, and optimizing for edge devices will support the deployment of scalable, effective, and user-centric cheating detection systems that uphold academic integrity in diverse online education settings.</p>
</sec>
</body>
<back>
<ack>
<p>We are grateful to the Princess Nourah bint Abdulrahman University Researchers Supporting Project number (PNURSP2025R752), Princess Nourah bint Abdulrahman University, Riyadh, Saudi Arabia. This paper also derived from a research grant funded by the Research, Development, and Innovation Authority (RDIA)&#x2014;Kingdom of Saudi Arabia&#x2014;with grant number (13325-psu-2023-PSNU-R-3-1-EF). Additionally, the authors would like to thank Prince Sultan University for their support.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>This work was funded by the Princess Nourah bint Abdulrahman University Researchers Supporting Project number (PNURSP2025R752), Princess Nourah bint Abdulrahman University, Riyadh, Saudi Arabia.</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>Conceptualization, Siham Essahraui and Ismail Lamaakal; data curation, Siham Essahraui and Ismail Lamaakal; formal analysis, Siham Essahraui, Ismail Lamaakal, Khalid El Makkaoui and Yassine Maleh; funding acquisition, May Almousa, Ali Abdullah S. AlQahtani and Ahmed A. Abd El-Latif; methodology, Siham Essahraui, Ismail Lamaakal, Khalid El Makkaoui, Mouncef Filali Bouami, Ibrahim Ouahbi and Yassine Maleh; project administration, Siham Essahraui, Ismail Lamaakal, Khalid El Makkaoui, Yassine Maleh, Ibrahim Ouahbi, May Almousa, Ali Abdullah S. AlQahtani, Ahmed A. Abd El-Latif and Mouncef Filali Bouami; software, Siham Essahraui and Ismail Lamaakal; supervision, Yassine Maleh, Khalid El Makkaoui, Ibrahim Ouahbi, Mouncef Filali Bouami, May Almousa, Ali Abdullah S. AlQahtani and Ahmed A. Abd El-Latif; validation, Siham Essahraui, Ismail Lamaakal, Khalid El Makkaoui, Ibrahim Ouahbi, Mouncef Filali Bouami and Yassine Maleh; visualization, Siham Essahraui and Ismail Lamaakal; writing&#x2014;original draft, Siham Essahraui and Ismail Lamaakal; writing&#x2014;review and editing, Yassine Maleh, Khalid El Makkaoui, Ibrahim Ouahbi, May Almousa, Mouncef Filali Bouami, Ali Abdullah S. AlQahtani and Ahmed A. Abd El-Latif. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>Data are contained within the article.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>French</surname> <given-names>S</given-names></string-name>, <string-name><surname>Dickerson</surname> <given-names>A</given-names></string-name>, <string-name><surname>Mulder</surname> <given-names>RA</given-names></string-name></person-group>. <article-title>A review of the benefits and drawbacks of high-stakes final examinations in higher education</article-title>. <source>Higher Educ</source>. <year>2024</year>;<volume>88</volume>(<issue>3</issue>):<fpage>893</fpage>&#x2013;<lpage>918</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10734-023-01148-z</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cui</surname> <given-names>X</given-names></string-name>, <string-name><surname>Du</surname> <given-names>C</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Impact of gamified learning experience on online learning effectiveness</article-title>. <source>IEEE Trans Learn Technol</source>. <year>2024</year>;<volume>17</volume>(<issue>2</issue>):<fpage>2076</fpage>&#x2013;<lpage>85</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TLT.2024.3462892</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ortiz-Lopez</surname> <given-names>A</given-names></string-name>, <string-name><surname>Olmos-Miguelanez</surname> <given-names>S</given-names></string-name>, <string-name><surname>Sanchez-Prieto</surname> <given-names>JC</given-names></string-name></person-group>. <article-title>Toward a new educational reality: a mapping review of the role of e-assessment in the new digital context</article-title>. <source>Educ Inf Technol</source>. <year>2024</year>;<volume>29</volume>(<issue>6</issue>):<fpage>7053</fpage>&#x2013;<lpage>80</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10639-023-12117-5</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Okoye</surname> <given-names>K</given-names></string-name>, <string-name><surname>Daruich</surname> <given-names>SDN</given-names></string-name>, <string-name><surname>De La</surname> <given-names>OJFE</given-names></string-name>, <string-name><surname>Casta&#x00F1;o</surname> <given-names>R</given-names></string-name>, <string-name><surname>Escamilla</surname> <given-names>J</given-names></string-name>, <string-name><surname>Hosseini</surname> <given-names>S</given-names></string-name></person-group>. <article-title>A text mining and statistical approach for assessment of pedagogical impact of students&#x2019; evaluation of teaching and learning outcome in education</article-title>. <source>IEEE Access</source>. <year>2023</year>;<volume>11</volume>(<issue>2</issue>):<fpage>9577</fpage>&#x2013;<lpage>96</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ACCESS.2023.3239779</pub-id>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Pongchaikul</surname> <given-names>P</given-names></string-name>, <string-name><surname>Vivithanaporn</surname> <given-names>P</given-names></string-name>, <string-name><surname>Somboon</surname> <given-names>N</given-names></string-name>, <string-name><surname>Tantasiri</surname> <given-names>J</given-names></string-name>, <string-name><surname>Suwanlikit</surname> <given-names>T</given-names></string-name>, <string-name><surname>Sukkul</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Remote online-exam dishonesty in medical school during COVID-19 pandemic: a real-world case with a multi-method approach</article-title>. <source>J Acad Ethics</source>. <year>2024</year>;<volume>22</volume>(<issue>3</issue>):<fpage>539</fpage>&#x2013;<lpage>59</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10805-024-09571-2</pub-id>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Rodr&#x00ED;guez-Paz</surname> <given-names>MX</given-names></string-name>, <string-name><surname>Gonz&#x00E1;lez-Mendivil</surname> <given-names>JA</given-names></string-name>, <string-name><surname>Z&#x00E1;rate-Garc&#x00ED;a</surname> <given-names>JA</given-names></string-name>, <string-name><surname>Zamora-Hern&#x00E1;ndez</surname> <given-names>I</given-names></string-name>, <string-name><surname>Nolazco-Flores</surname> <given-names>JA</given-names></string-name></person-group>. <article-title>A hybrid teaching model for engineering courses suitable for pandemic conditions</article-title>. <source>IEEE Rev Iberoam Tecnol Aprendiz</source>. <year>2021</year>;<volume>16</volume>(<issue>3</issue>):<fpage>267</fpage>&#x2013;<lpage>75</lpage>. doi:<pub-id pub-id-type="doi">10.1109/RITA.2021.3122893</pub-id>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Newton</surname> <given-names>PM</given-names></string-name>, <string-name><surname>Essex</surname> <given-names>K</given-names></string-name></person-group>. <article-title>How common is cheating in online exams and did it increase during the COVID-19 pandemic? A systematic review</article-title>. <source>J Acad Ethics</source>. <year>2024</year>;<volume>22</volume>(<issue>2</issue>):<fpage>323</fpage>&#x2013;<lpage>43</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10805-023-09485-5</pub-id>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>J</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Bai</surname> <given-names>X</given-names></string-name>, <string-name><surname>Kaur</surname> <given-names>R</given-names></string-name>, <string-name><surname>Xia</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Multiple instance learning for cheating detection and localization in online examinations</article-title>. <source>IEEE Trans Cogn Dev Syst</source>. <year>2024</year>;<volume>16</volume>(<issue>4</issue>):<fpage>1315</fpage>&#x2013;<lpage>26</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TCDS.2024.3349705</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ryu</surname> <given-names>R</given-names></string-name>, <string-name><surname>Yeom</surname> <given-names>S</given-names></string-name>, <string-name><surname>Herbert</surname> <given-names>D</given-names></string-name>, <string-name><surname>Dermoudy</surname> <given-names>J</given-names></string-name></person-group>. <article-title>A comprehensive survey of context-aware continuous implicit authentication in online learning environments</article-title>. <source>IEEE Access</source>. <year>2023</year>;<volume>11</volume>(<issue>1</issue>):<fpage>24561</fpage>&#x2013;<lpage>73</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ACCESS.2023.3253484</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Susnjak</surname> <given-names>T</given-names></string-name>, <string-name><surname>McIntosh</surname> <given-names>TR</given-names></string-name></person-group>. <article-title>ChatGPT: the end of online exam integrity?</article-title> <source>Educ Sci</source>. <year>2024</year>;<volume>14</volume>(<issue>6</issue>):<fpage>656</fpage>. doi:<pub-id pub-id-type="doi">10.3390/educsci14060656</pub-id>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Taherkhani</surname> <given-names>R</given-names></string-name>, <string-name><surname>Aref</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Students&#x2019; online cheating reasons and strategies: EFL teachers&#x2019; strategies to abolish cheating in online examinations</article-title>. <source>J Acad Ethics</source>. <year>2024</year>;<volume>22</volume>(<issue>22</issue>):<fpage>539</fpage>&#x2013;<lpage>59</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10805-024-09502-1</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Dilini</surname> <given-names>N</given-names></string-name>, <string-name><surname>Senaratne</surname> <given-names>A</given-names></string-name>, <string-name><surname>Yasarathna</surname> <given-names>T</given-names></string-name>, <string-name><surname>Warnajith</surname> <given-names>N</given-names></string-name>, <string-name><surname>Seneviratne</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Cheating detection in browser-based online exams through eye gaze tracking</article-title>. In: <conf-name>Proceedings of the 6th International Conference on Information Technology Research (ICITR); 2021 Dec 1&#x2013;2</conf-name>; <publisher-loc>Colombo, Sri Lanka. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2021</year>. p. <fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICITR54349.2021.9657277</pub-id>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Khabbachi</surname> <given-names>I</given-names></string-name>, <string-name><surname>Mdaghri-Alaoui</surname> <given-names>G</given-names></string-name>, <string-name><surname>Zouhair</surname> <given-names>A</given-names></string-name>, <string-name><surname>Mahboub</surname> <given-names>A</given-names></string-name></person-group>. <article-title>A cheating detection system in online exams through real-time facial emotion recognition of students</article-title>. In: <conf-name>Proceedings of the 2024 International Conference on Computer, Internet of Things and Microwave Systems (ICCIMS); 2024 Jul 29&#x2013;31</conf-name>; <publisher-loc>Gatineau, QC, Canada. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2024</year>. p. <fpage>1</fpage>&#x2013;<lpage>5</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICCIMS61672.2024.10690812</pub-id>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mellar</surname> <given-names>H</given-names></string-name>, <string-name><surname>Peytcheva-Forsyth</surname> <given-names>R</given-names></string-name>, <string-name><surname>Kocdar</surname> <given-names>S</given-names></string-name>, <string-name><surname>Karadeniz</surname> <given-names>A</given-names></string-name>, <string-name><surname>Yovkova</surname> <given-names>B</given-names></string-name></person-group>. <article-title>Addressing cheating in e-assessment using student authentication and authorship checking systems: teachers&#x2019; perspectives</article-title>. <source>Int J Educ Integr</source>. <year>2018</year>;<volume>14</volume>(<issue>1</issue>):<fpage>1</fpage>&#x2013;<lpage>21</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s40979-018-0025-x</pub-id>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Verma</surname> <given-names>P</given-names></string-name>, <string-name><surname>Malhotra</surname> <given-names>N</given-names></string-name>, <string-name><surname>Suri</surname> <given-names>R</given-names></string-name>, <string-name><surname>Kumar</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Automated smart artificial intelligence-based proctoring system using deep learning</article-title>. <source>Soft Comput</source>. <year>2024</year>;<volume>28</volume>(<issue>4</issue>):<fpage>3479</fpage>&#x2013;<lpage>89</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s00500-023-08696-7</pub-id>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chatterjee</surname> <given-names>P</given-names></string-name>, <string-name><surname>Dansana</surname> <given-names>J</given-names></string-name>, <string-name><surname>Swain</surname> <given-names>S</given-names></string-name>, <string-name><surname>Gourisaria</surname> <given-names>MK</given-names></string-name>, <string-name><surname>Bandyopadhyay</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Identity verification in real-time proctoring: an integrated approach with face recognition and eye tracking</article-title>. In: <conf-name>Proceedings of the 2024 International Conference on Intelligent Algorithms and Computational Intelligence Systems (IACIS); 2024 Aug 23&#x2013;24</conf-name>; <publisher-loc>Hassan, India. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2024</year>. p. <fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1109/IACIS61494.2024.10721819</pub-id>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ramzan</surname> <given-names>M</given-names></string-name>, <string-name><surname>Abid</surname> <given-names>A</given-names></string-name>, <string-name><surname>Bilal</surname> <given-names>M</given-names></string-name>, <string-name><surname>Aamir</surname> <given-names>KM</given-names></string-name>, <string-name><surname>Memon</surname> <given-names>SA</given-names></string-name>, <string-name><surname>Chung</surname> <given-names>TS</given-names></string-name></person-group>. <article-title>Effectiveness of pre-trained CNN networks for detecting abnormal activities in online exams</article-title>. <source>IEEE Access</source>. <year>2024</year>;<volume>81</volume>(<issue>17</issue>):<fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1109/IACIS61494.2024.10721819</pub-id>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kaddoura</surname> <given-names>S</given-names></string-name>, <string-name><surname>Gumaei</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Towards effective and efficient online exam systems using deep learning-based cheating detection approach</article-title>. <source>Intell Syst Appl</source>. <year>2022</year>;<volume>16</volume>:<fpage>200153</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.iswa.2022.200153</pub-id>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Nurpeisova</surname> <given-names>A</given-names></string-name>, <string-name><surname>Shaushenova</surname> <given-names>A</given-names></string-name>, <string-name><surname>Mutalova</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Ongarbayeva</surname> <given-names>M</given-names></string-name>, <string-name><surname>Niyazbekova</surname> <given-names>S</given-names></string-name>, <string-name><surname>Bekenova</surname> <given-names>A</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Research on the development of a proctoring system for conducting online exams in Kazakhstan</article-title>. <source>Computation</source>. <year>2023</year>;<volume>11</volume>(<issue>6</issue>):<fpage>120</fpage>. doi:<pub-id pub-id-type="doi">10.3390/computation11060120</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ozdamli</surname> <given-names>F</given-names></string-name>, <string-name><surname>Aljarrah</surname> <given-names>A</given-names></string-name>, <string-name><surname>Karagozlu</surname> <given-names>D</given-names></string-name>, <string-name><surname>Ababneh</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Facial recognition system to detect student emotions and cheating in distance learning</article-title>. <source>Sustainability</source>. <year>2022</year>;<volume>14</volume>(<issue>20</issue>):<fpage>13230</fpage>. doi:<pub-id pub-id-type="doi">10.3390/su142013230</pub-id>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Jing</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>G</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Multi-perspective adaptive paperless examination cheating detection system based on image recognition</article-title>. <source>Appl Sci</source>. <year>2024</year>;<volume>14</volume>(<issue>10</issue>):<fpage>4048</fpage>. doi:<pub-id pub-id-type="doi">10.3390/app14104048</pub-id>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dang</surname> <given-names>TL</given-names></string-name>, <string-name><surname>Hoang</surname> <given-names>NMN</given-names></string-name>, <string-name><surname>Nguyen</surname> <given-names>TV</given-names></string-name>, <string-name><surname>Nguyen</surname> <given-names>HV</given-names></string-name>, <string-name><surname>Dang</surname> <given-names>QM</given-names></string-name>, <string-name><surname>Tran</surname> <given-names>QH</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Auto-proctoring using computer vision in MOOCs system</article-title>. <source>Multimed Tools Appl</source>. <year>2024</year>;<volume>1</volume>(<issue>7</issue>):<fpage>1</fpage>&#x2013;<lpage>27</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s11042-024-20099-w</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Potluri</surname> <given-names>T</given-names></string-name>, <string-name><surname>Venkatramaphanikumar</surname> <given-names>S</given-names></string-name>, <string-name><surname>Kolli</surname> <given-names>VKK</given-names></string-name></person-group>. <article-title>An automated online proctoring system using Attentive-Net to assess student mischievous behavior</article-title>. <source>Multimed Tools Appl</source>. <year>2023</year>;<volume>82</volume>(<issue>20</issue>):<fpage>30375</fpage>&#x2013;<lpage>404</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s11042-023-14604-w</pub-id>; <pub-id pub-id-type="pmid">36846528</pub-id></mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>X</given-names></string-name>, <string-name><surname>Li</surname> <given-names>H</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Double-frame rate-based cheat detection</article-title>. In: <conf-name>Third International Conference on Computer Vision and Pattern Analysis (ICCPA 2023)</conf-name>; <publisher-loc>Hangzhou, China</publisher-loc>: <publisher-name>SPIE</publisher-name>; <year>2023</year>. Vol. <volume>12754</volume>. p. <fpage>529</fpage>&#x2013;<lpage>35</lpage>. doi:<pub-id pub-id-type="doi">10.1117/12.2684283</pub-id>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Thampan</surname> <given-names>N</given-names></string-name>, <string-name><surname>Arumugam</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Smart online exam invigilation using AI-based facial detection and recognition algorithms</article-title>. In: <conf-name>Proceedings of the 2022 2nd Asian Conference on Innovation in Technology (ASIANCON); 2022 Aug 26&#x2013;28</conf-name>; <publisher-loc>Ravet, India. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2022</year>. p. <fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ASIANCON55314.2022.9908748</pub-id>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Singh</surname> <given-names>T</given-names></string-name>, <string-name><surname>Nair</surname> <given-names>RR</given-names></string-name>, <string-name><surname>Babu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Duraisamy</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Enhancing academic integrity in online assessments: introducing an effective online exam proctoring model using YOLO</article-title>. <source>Procedia Comput Sci</source>. <year>2024</year>;<volume>235</volume>(<issue>12</issue>):<fpage>1399</fpage>&#x2013;<lpage>408</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.procs.2024.04.131</pub-id>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Roy</surname> <given-names>K</given-names></string-name>, <string-name><surname>Chanda</surname> <given-names>D</given-names></string-name></person-group>. <article-title>A robust webcam-based eye gaze estimation system for human-computer interaction</article-title>. In: <conf-name>Proceedings of the 2022 International Conference on Innovation in Science and Engineering Technology (ICISET); 2022 Feb 26&#x2013;27</conf-name>; <publisher-loc>Chittagong, Bangladesh. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2022</year>. p. <fpage>146</fpage>&#x2013;<lpage>51</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICISET54810.2022.9775896</pub-id>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Parkhi</surname> <given-names>PN</given-names></string-name>, <string-name><surname>Patel</surname> <given-names>A</given-names></string-name>, <string-name><surname>Solanki</surname> <given-names>D</given-names></string-name>, <string-name><surname>Ganwani</surname> <given-names>H</given-names></string-name>, <string-name><surname>Anandani</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Proficient exam monitoring system using deep learning techniques</article-title>. In: <conf-name>International Conference on Information and Communication Technology for Competitive Strategies</conf-name>; <publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer Nature</publisher-name>; <year>2022</year>. p. <fpage>31</fpage>&#x2013;<lpage>49</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-981-97-0744-7_3</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Gadkar</surname> <given-names>S</given-names></string-name>, <string-name><surname>Vora</surname> <given-names>B</given-names></string-name>, <string-name><surname>Chotai</surname> <given-names>K</given-names></string-name>, <string-name><surname>Lakhani</surname> <given-names>S</given-names></string-name>, <string-name><surname>Katudia</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Online examination auto-proctoring system</article-title>. In: <conf-name>Proceedings of the 2023 International Conference on Advanced Computing Technologies and Applications (ICACTA); 2023 Oct 6&#x2013;7</conf-name>; <publisher-loc>Mumbai, India. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2023</year>. p. <fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICACTA58201.2023.10392679</pub-id>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Atabay</surname> <given-names>HA</given-names></string-name>, <string-name><surname>Hassanpour</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Semi-supervised anomaly detection in electronic-exam proctoring based on skeleton similarity</article-title>. In: <conf-name>Proceedings of the 2023 11th European Workshop on Visual Information Processing (EUVIP); 2023 Sep 11&#x2013;14</conf-name>; <publisher-loc>Gjovik, Norway. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2023</year>. p. <fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1109/EUVIP58404.2023.10323052</pub-id>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Samir</surname> <given-names>MA</given-names></string-name>, <string-name><surname>Maged</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Atia</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Exam cheating detection system with multiple-human pose estimation</article-title>. In: <conf-name>Proceedings of the 2021 IEEE International Conference on Computing (ICOCO); 2021 Nov 17&#x2013;19</conf-name>; <publisher-loc>Kuala Lumpur, Malaysia. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2021</year>. p. <fpage>236</fpage>&#x2013;<lpage>40</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICOCO53166.2021.9673534</pub-id>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Atoum</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>AX</given-names></string-name>, <string-name><surname>Hsu</surname> <given-names>SDH</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>X</given-names></string-name></person-group>. <article-title>Automated online exam proctoring</article-title>. <source>IEEE Trans Multimed</source>. <year>2017</year>;<volume>19</volume>(<issue>7</issue>):<fpage>1609</fpage>&#x2013;<lpage>24</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TMM.2017.2656064</pub-id>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tejaswi</surname> <given-names>P</given-names></string-name>, <string-name><surname>Venkatramaphanikumar</surname> <given-names>S</given-names></string-name>, <string-name><surname>Kishore</surname> <given-names>KVK</given-names></string-name></person-group>. <article-title>Proctor Net: an AI framework for suspicious activity detection in online proctored examinations</article-title>. <source>Measurement</source>. <year>2023</year>;<volume>206</volume>(<issue>1</issue>):<fpage>112266</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.measurement.2022.112266</pub-id>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>LeCun</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Bengio</surname> <given-names>Y</given-names></string-name></person-group>. <chapter-title>Convolutional networks for images, speech, and time series</chapter-title>. In: <person-group person-group-type="editor"><string-name><surname>Arbib</surname> <given-names>MA</given-names></string-name></person-group>, editor. <source>The handbook of brain theory and neural networks</source>. <edition>1st</edition> ed. <publisher-loc>Cambridge, MA, USA</publisher-loc>: <publisher-name>MIT Press</publisher-name>; <year>1995</year>. p. <fpage>3361</fpage>&#x2013;<lpage>70</lpage>. doi:<pub-id pub-id-type="doi">10.5555/303568.303704</pub-id>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Howard</surname> <given-names>AG</given-names></string-name></person-group>. <article-title>MobileNets: efficient convolutional neural networks for mobile vision applications.arXiv:1704.04861</article-title>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.1704.04861</pub-id>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>He</surname> <given-names>K</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>S</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Deep residual learning for image recognition</article-title>. In: <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</conf-name>; <year>2016</year>. p. <fpage>770</fpage>&#x2013;<lpage>8</lpage>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.1512.03385</pub-id>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Tan</surname> <given-names>M</given-names></string-name>, <string-name><surname>Le</surname> <given-names>Q</given-names></string-name></person-group>. <article-title>EfficientNet: rethinking model scaling for convolutional neural networks</article-title>. In: <conf-name>Proceedings of the 36th International Conference on Machine Learning (ICML); 2019 Jun 9&#x2013;15</conf-name>; <publisher-loc>Long Beach, CA, USA</publisher-loc>. p. <fpage>6105</fpage>&#x2013;<lpage>14</lpage>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.1905.11946</pub-id>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Mao</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>CY</given-names></string-name>, <string-name><surname>Feichtenhofer</surname> <given-names>C</given-names></string-name>, <string-name><surname>Darrell</surname> <given-names>T</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>S</given-names></string-name></person-group>. <article-title>A ConvNet for the 2020s</article-title>. In: <conf-name>Proceedings of the 2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2022 Jun 18&#x2013;24</conf-name>; <publisher-name>New Orleans, LA, USA. Piscataway, NJ</publisher-name>: <publisher-name>IEEE</publisher-name>; <year>2022</year>. p. <fpage>11976</fpage>&#x2013;<lpage>86</lpage>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2201.03545</pub-id>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Squeeze-and-excitation networks</article-title>. In: <conf-name>Proceedings of the 2018 IEEE Conference on Computer Vision and Pattern Recognition (CVPR); 2018 Jun 18&#x2013;22</conf-name>; <publisher-loc>Salt Lake City, UT, USA. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2018</year>. p. <fpage>7132</fpage>&#x2013;<lpage>41</lpage>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.1709.01507</pub-id>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Woo</surname> <given-names>S</given-names></string-name>, <string-name><surname>Park</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>JY</given-names></string-name>, <string-name><surname>Kweon</surname> <given-names>IS</given-names></string-name></person-group>. <article-title>CBAM: convolutional block attention module</article-title>. In: <conf-name>Proceedings of the 15th European Conference on Computer Vision (ECCV); 2018 Sep 8&#x2013;14</conf-name>; <publisher-loc>Munich, Germany. Cham</publisher-loc>: <publisher-name>Springer</publisher-name>; <year>2018</year>. p. <fpage>3</fpage>&#x2013;<lpage>19</lpage>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.1807.06521</pub-id>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Swin Transformer: hierarchical vision transformer using shifted windows</article-title>. In: <conf-name>Proceedings of the 2021 IEEE/CVF International Conference on Computer Vision (ICCV); 2021 Oct 10&#x2013;17</conf-name>; <publisher-loc>Montreal, QC, Canada. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2021</year>. p. <fpage>10012</fpage>&#x2013;<lpage>22</lpage>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2103.14030</pub-id>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Redmon</surname> <given-names>J</given-names></string-name></person-group>. <article-title>You only look once: unified, real-time object detection</article-title>. In: <conf-name>Proceedings of the 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR); 2016 Jun 27&#x2013;30</conf-name>; <publisher-loc>Las Vegas, NV, USA. Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2016</year>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2016.91</pub-id>.</mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Jing</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Research on remote intelligent monitoring system for online examination based on gaze direction classification</article-title>. <source>Proc 9th Int Conf Comput Artif Intell</source>. <year>2023</year>;<fpage>535</fpage>&#x2013;<lpage>40</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3594315.3594369</pub-id>.</mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Huang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Yin</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>H</given-names></string-name>, <string-name><surname>Chai</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ouyang</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Video-based recognition of online learning behaviors using attention mechanisms</article-title>. In: <conf-name>2024 IEEE International Conference on Teaching, Assessment and Learning for Engineering (TALE)</conf-name>; <publisher-loc>Bengaluru, India</publisher-loc>; <year>2024</year>. p. <fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TALE62452.2024.10834376</pub-id>.</mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Malhotra</surname> <given-names>N</given-names></string-name>, <string-name><surname>Suri</surname> <given-names>R</given-names></string-name>, <string-name><surname>Verma</surname> <given-names>P</given-names></string-name>, <string-name><surname>Kumar</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Smart artificial intelligence based online proctoring system</article-title>. In: <conf-name>2022 IEEE Delhi section conference (DELCON 2022)</conf-name>; <publisher-loc>New Delhi, India</publisher-loc>: <publisher-name>IEEE</publisher-name>; <year>2022</year>. p. <fpage>1</fpage>&#x2013;<lpage>5</lpage>. doi:<pub-id pub-id-type="doi">10.1109/DELCON54057.2022.9753313</pub-id>.</mixed-citation></ref>
</ref-list>
</back></article>