<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMES</journal-id>
<journal-id journal-id-type="nlm-ta">CMES</journal-id>
<journal-id journal-id-type="publisher-id">CMES</journal-id>
<journal-title-group>
<journal-title>Computer Modeling in Engineering &#x0026; Sciences</journal-title>
</journal-title-group>
<issn pub-type="epub">1526-1506</issn>
<issn pub-type="ppub">1526-1492</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">61744</article-id>
<article-id pub-id-type="doi">10.32604/cmes.2025.061744</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Computational Optimization of RIS-Enhanced Backscatter and Direct Communication for 6G IoT: A DDPG-Based Approach with Physical Layer Security</article-title>
<alt-title alt-title-type="left-running-head">Computational Optimization of RIS-Enhanced Backscatter and Direct Communication for 6G IoT: A DDPG-Based Approach with Physical Layer Security</alt-title>
<alt-title alt-title-type="right-running-head">Computational Optimization of RIS-Enhanced Backscatter and Direct Communication for 6G IoT: A DDPG-Based Approach with Physical Layer Security</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Abideen</surname><given-names>Syed Zain Ul</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Kamal</surname><given-names>Mian Muhammad</given-names></name><xref ref-type="aff" rid="aff-2">2</xref><email>mmkamal@seu.edu.cn</email></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Alharbi</surname><given-names>Eaman</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Malik</surname><given-names>Ashfaq Ahmad</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Alhalabi</surname><given-names>Wadee</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-6" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Anwar</surname><given-names>Muhammad Shahid</given-names></name><xref ref-type="aff" rid="aff-6">6</xref><email>shahidanwar786@gachon.ac.kr</email></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Ali</surname><given-names>Liaqat</given-names></name><xref ref-type="aff" rid="aff-7">7</xref></contrib>
<aff id="aff-1"><label>1</label><institution>College of Computer Science and Technology, Qingdao University</institution>, <addr-line>Qingdao, 266071</addr-line>, <country>China</country></aff>
<aff id="aff-2"><label>2</label><institution>School of Electronic Science and Engineering, Southeast University</institution>, <addr-line>Nanjing, 210018</addr-line>, <country>China</country></aff>
<aff id="aff-3"><label>3</label><institution>Computer Science Department, Faculty of Computing and Information Technology, King Abdulaziz University</institution>, <addr-line>Jeddah, 80200</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-4"><label>4</label><institution>Department of Quality Assurance, Al-Kawthar University</institution>, <addr-line>Karachi, 75300</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-5"><label>5</label><institution>Department of Computer Science, Immersive Virtual Reality Research Group, King Abdulaziz University</institution>, <addr-line>Jeddah, 80200</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-6"><label>6</label><institution>Department of AI and Software, Gachon University</institution>, <addr-line>Seongnam-si, 13120</addr-line>, <country>Republic of Korea</country></aff>
<aff id="aff-7"><label>7</label><institution>Department of Electrical Engineering, University of Science and Technology</institution>, <addr-line>Bannu, 28100</addr-line>, <country>Pakistan</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Authors: Mian Muhammad Kamal. Email: <email>mmkamal@seu.edu.cn</email>; Muhammad Shahid Anwar. Email: <email>shahidanwar786@gachon.ac.kr</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2025</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>03</day><month>03</month><year>2025</year>
</pub-date>
<volume>142</volume>
<issue>3</issue>
<fpage>2191</fpage>
<lpage>2210</lpage>
<history>
<date date-type="received">
<day>02</day>
<month>12</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>2</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 The Authors.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMES_61744.pdf"></self-uri>
<abstract>
<p>The rapid evolution of wireless technologies and the advent of 6G networks present new challenges and opportunities for Internet of Things (IoT) applications, particularly in terms of ultra-reliable, secure, and energy-efficient communication. This study explores the integration of Reconfigurable Intelligent Surfaces (RIS) into IoT networks to enhance communication performance. Unlike traditional passive reflector-based approaches, RIS is leveraged as an active optimization tool to improve both backscatter and direct communication modes, addressing critical IoT challenges such as energy efficiency, limited communication range, and double-fading effects in backscatter communication. We propose a novel computational framework that combines RIS functionality with Physical Layer Security (PLS) mechanisms, optimized through the algorithm known as Deep Deterministic Policy Gradient (DDPG). This framework adaptively adapts RIS configurations and transmitter beamforming to reduce key challenges, including imperfect channel state information (CSI) and hardware limitations like quantized RIS phase shifts. By optimizing both RIS settings and beamforming in real-time, our approach outperforms traditional methods by significantly increasing secrecy rates, improving spectral efficiency, and enhancing energy efficiency. Notably, this framework adapts more effectively to the dynamic nature of wireless channels compared to conventional optimization techniques, providing scalable solutions for large-scale RIS deployments. Our results demonstrate substantial improvements in communication performance setting a new benchmark for secure, efficient and scalable 6G communication. This work offers valuable insights for the future of IoT networks, with a focus on computational optimization, high spectral efficiency and energy-aware operations.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Computational optimization</kwd>
<kwd>reconfigurable intelligent surfaces (RIS)</kwd>
<kwd>6G networks</kwd>
<kwd>IoT and DDPG</kwd>
<kwd>physical layer security (PLS)</kwd>
<kwd>backscatter communication</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>deanship of scientific research (DSR), King Abdukaziz University, Jeddah, under grant No</funding-source>
<award-id>G-1436-611&#x2013;225</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>The dynamic advancement in wireless technology is driving the transition toward sixth-generation (6G) networks which aim to meet the demands of revolutionary data rates with minimal latency and secure communication for emerging Internet of Things (IoT) applications. With IoT devices proliferating across diverse industries, the networks face critical challenges in scalability, bandwidth efficiency and security [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-2">2</xref>]. Addressing these difficulties requires innovative communication frameworks that can efficiently manage resources while safeguarding against security threats.</p>
<p>BackCom is increasingly recognized as a groundbreaking energy efficient solution by reflecting incident signals from a carrier emitter without generating new radio waves [<xref ref-type="bibr" rid="ref-3">3</xref>,<xref ref-type="bibr" rid="ref-4">4</xref>]. This technique is particularly attractive for resource constrained IoT devices, as it significantly reduces power consumption. However, BackCom suffers from inherent limitations such as signal degradation caused by double-fading effects and restricted communication range, which hinder its scalability and practical deployment in dense IoT networks [<xref ref-type="bibr" rid="ref-5">5</xref>]. Overcoming these drawbacks requires advanced technologies capable of optimizing signal propagation and enhancing communication reliability.</p>
<p>Reconfigurable Intelligent Surface (RIS) have recently emerged as a transformative technology in wireless networks [<xref ref-type="bibr" rid="ref-6">6</xref>]. RIS is composed of an extensive array of passive reflective elements. capable of adaptively manipulating the phase and magnitude of approaching electromagnetic waves. This ability makes RIS a powerful enabler for enhancing both backscatter and direct communication by improving signal propagation, mitigating interference and boosting spectral efficiency in complex IoT environments [<xref ref-type="bibr" rid="ref-7">7</xref>]. Unlike conventional systems that treat the environment as a passive channel, RIS-enabled systems actively shape the propagation environment enabling intelligent signal control and reducing the impact of path loss [<xref ref-type="bibr" rid="ref-8">8</xref>].</p>
<p>To fully exploit the transformative potential of RIS, this study introduces an innovative RIS-aided Joint Backscattering and Communication (JBAC) system setting a new benchmark for 6G networks. The proposed system seamlessly integrates RIS with traditional backscatter communication, addressing inherent limitations by enhancing transmission efficiency and optimizing spectral utilization [<xref ref-type="bibr" rid="ref-9">9</xref>]. Beyond boosting communication performance, the study prioritizes data privacy and security by embedding PLS mechanisms [<xref ref-type="bibr" rid="ref-10">10</xref>]. By leveraging the inherent randomness of wireless channels, the PLS provides robust protection against eavesdropping serving as a lightweight, energy efficient alternative to conventional encryption techniques [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-12">12</xref>].</p>
<p>To achieve real-time optimization the system employs the DDPG algorithm a reinforcement learning method well-suited for continuous action spaces [<xref ref-type="bibr" rid="ref-13">13</xref>]. DDPG dynamically optimizes RIS phase shifts and transmitter beamforming, enabling adaptive responses to changing channel conditions. By seamlessly coordinating RIS configurations and power allocation at the transmitter, the proposed framework enhances communication performance while strengthening security measures [<xref ref-type="bibr" rid="ref-14">14</xref>]. Furthermore, the integration of reinforcement learning ensures robust operation in the context of practical issues like incomplete channel state information (CSI) and RIS phase shift quantization</p>
<p>This work combines RIS, PLS and reinforcement learning to address the critical challenges of security and efficiency in IoT networks. By synergizing these advanced technologies the proposed framework sets new benchmarks for 6G communication protocols. It achieves significant improvements in secrecy rate and transmission efficiency, enabling the development of interconnected, secure and energy-conscious IoT ecosystems. The results of this research contribute substantially to the growing body of knowledge around RIS-enabled systems providing a foundation for high-performance IoT networks in the 6G era.</p>
<sec id="s1_1">
<label>1.1</label>
<title>Related Work</title>
<p>The emergence of RIS technology has redefined the concept of wireless communication by enabling the control of electromagnetic waves through programmable surfaces, as summarized in <xref ref-type="table" rid="table-1">Table 1</xref>, which dynamically manipulate phase shifts and amplitudes [<xref ref-type="bibr" rid="ref-15">15</xref>,<xref ref-type="bibr" rid="ref-16">16</xref>]. RIS technology is increasingly being integrated into 6G communication systems to meet the demands of high transmission rates, incredibly low latency and ubiquitous connectivity especially for IoT applications [<xref ref-type="bibr" rid="ref-17">17</xref>]. Through RIS, communication systems can reduce interference, improve spectral efficiency, and mitigate fading effects, thereby extending the operational range and robustness of both backscatter communication (BackCom) and direct transmissions [<xref ref-type="bibr" rid="ref-18">18</xref>,<xref ref-type="bibr" rid="ref-19">19</xref>].</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Comparison of related work</title>
</caption>
<table>
<colgroup>
<col/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th>Study</th>
<th align="center">Focus area</th>
<th align="center">Strengths</th>
<th align="center">Limitations</th>
<th align="center">Proposed framework advantage</th>
</tr>
</thead>
<tbody>
<tr>
<td>Pan et al. [<xref ref-type="bibr" rid="ref-15">15</xref>]</td>
<td>RIS for spectral efficiency</td>
<td>High spectral efficiency and reduced fading</td>
<td>No focus on security or BackCom integration</td>
<td>Joint optimization of RIS and BackCom</td>
</tr>
<tr>
<td>Jiang et al. [<xref ref-type="bibr" rid="ref-3">3</xref>]</td>
<td>BackCom for IoT</td>
<td>Energy efficiency for IoT devices</td>
<td>Double-fading effects reduce reliability</td>
<td>Mitigates double-fading via RIS</td>
</tr>
<tr>
<td>Nguyen et al. [<xref ref-type="bibr" rid="ref-26">26</xref>]</td>
<td>RIS-aided PLS</td>
<td>Improved secrecy capacity using RIS</td>
<td>Assumes perfect CSI and continuous phases</td>
<td>Robust to imperfect CSI and quantized phases</td>
</tr>
<tr>
<td>Puspitasari et al. [<xref ref-type="bibr" rid="ref-29">29</xref>]</td>
<td>ML for RIS optimization</td>
<td>Effective real-time optimization</td>
<td>Limited scalability to large-scale RIS</td>
<td>Scalable to large RIS configurations</td>
</tr>
<tr>
<td>Liang et al. [<xref ref-type="bibr" rid="ref-20">20</xref>]</td>
<td>Bistatic BackCom with RIS</td>
<td>Enhanced signal propagation and coverage</td>
<td>No integration with security mechanisms</td>
<td>Integrates PLS with RIS and BackCom</td>
</tr>
<tr>
<td><bold>Proposed work</bold></td>
<td>Joint RIS, BackCom, and PLS</td>
<td>Comprehensive optimization, real-time learning</td>
<td align ="center">&#x2013;</td>
<td>Dynamic, scalable, and secure solution</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Backscatter communication has gained prominence for its energy-efficient data transmission capabilities by using passive reflection of continuous carrier waves. This makes it particularly suitable for IoT devices with constrained energy resources [<xref ref-type="bibr" rid="ref-3">3</xref>]. However, BackCom faces key challenges such as the double-fading effect, which severely impacts the quality of transmission and limits communication range [<xref ref-type="bibr" rid="ref-20">20</xref>,<xref ref-type="bibr" rid="ref-21">21</xref>]. To overcome these limitations, researchers have proposed RIS-aided BackCom systems that enhance transmission efficiency by optimizing signal paths through intelligent reflections [<xref ref-type="bibr" rid="ref-22">22</xref>]. By introducing additional propagation channels, RIS helps mitigate path loss and extend coverage, facilitating seamless communication in dense IoT environments [<xref ref-type="bibr" rid="ref-23">23</xref>,<xref ref-type="bibr" rid="ref-24">24</xref>].</p>
<p>In parallel, the rise of PLS has been motivated by the growing need for lightweight security solutions in IoT and wireless systems. PLS takes advantage of the inherent randomness in wireless channels to secure data transmission without the computational overhead associated with cryptographic methods [<xref ref-type="bibr" rid="ref-25">25</xref>,<xref ref-type="bibr" rid="ref-12">12</xref>]. RIS-assisted PLS frameworks have demonstrated the ability to boost secrecy capacity by tailoring phase shifts to steer reflected signals toward legitimate users while weakening signals received by potential eavesdroppers [<xref ref-type="bibr" rid="ref-13">13</xref>,<xref ref-type="bibr" rid="ref-26">26</xref>]. This has led to the development of RIS-aided wiretap communication models where beamforming strategies are optimized to maximize secrecy rates [<xref ref-type="bibr" rid="ref-27">27</xref>,<xref ref-type="bibr" rid="ref-28">28</xref>]. These models ensure that confidential data remains protected even in highly dynamic network environments.</p>
<p>The adoption of machine learning (ML) algorithms for managing RIS configurations is another significant advancement. Reinforcement learning techniques, such as the DDPG algorithm, have been used to dynamically adjust RIS parameters and optimize beamforming patterns [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-12">12</xref>]. These algorithms are particularly effective in continuous action spaces such as those encountered in RIS systems where precise phase shifts are required to optimize both signal strength and security [<xref ref-type="bibr" rid="ref-30">30</xref>]. DDPG-based frameworks enhance the adaptability of RIS-aided communication systems by enabling real-time optimization in response to changing environmental conditions [<xref ref-type="bibr" rid="ref-31">31</xref>].</p>
<p>Further research has focused on the development of bistatic BackCom systems integrated with RIS. In these systems, the carrier emitter and receiver are spatially separated enabling more flexible network deployments [<xref ref-type="bibr" rid="ref-24">24</xref>]. RIS enhances these setups by introducing controllable propagation paths leading to better signal quality and reduced interference [<xref ref-type="bibr" rid="ref-20">20</xref>]. Resource allocation and beamforming optimization are critical research areas that empower RIS to achieve an effective balance between power distribution and the fulfillment of secrecy constraints [<xref ref-type="bibr" rid="ref-32">32</xref>].</p>
<p>As the number of RIS elements increases in advanced 6G communication systems, the computational complexity of optimization algorithms like DDPG becomes a critical consideration. This paper investigates how the complexity of DDPG scales with the number of RIS elements and demonstrates that the proposed approach remains efficient and scalable, even for large-scale RIS configurations.</p>
<p>Recent studies highlight the effectiveness of Deep Reinforcement Learning (DRL) in optimizing RIS-assisted networks, reference [<xref ref-type="bibr" rid="ref-33">33</xref>] proposed a HAP-based integrated satellite-aerial-terrestrial relay network using a DRL-based Long Short-Term Memory - Double Deep Q-Network (LSTM-DDQN) framework to maximize ergodic rate under Unmanned Aerial Vehicle (UAV) energy constraints. Similarly, reference [<xref ref-type="bibr" rid="ref-34">34</xref>] explored multi-RIS-assisted satellite-UAV-terrestrial networks, addressing dynamic environments and spectrum scarcity with a multi-objective deep deterministic policy gradient (MO-DDPG) algorithm to optimize achievable rate and energy efficiency. These works demonstrate DRL&#x2019;s potential for efficient RIS-enabled integrated networks.</p>
<p>As the role of RIS expands in the realm of 6G networks, several works have established new benchmarks for secrecy rate, spectral efficiency, and power management by integrating RIS, PLS, and advanced ML techniques. This study builds upon these foundations, proposing a novel Joint Backscattering and Communication (JBAC) system that leverages DDPG to dynamically optimize RIS configurations. The framework ensures efficient and secure communication by enhancing legitimate signal paths while suppressing those accessible to eavesdroppers, thereby aligning with the key goals of future 6G-enabled IoT networks.</p>
</sec>
<sec id="s1_2">
<label>1.2</label>
<title>Motivation and Contributions</title>
<p>The rapid evolution of wireless communication systems and the advent of 6G networks bring unprecedented opportunities for hyper-connected environments, primarily driven by the proliferation of IoT devices. However, these advancements also pose significant challenges: ensuring secure, efficient, and reliable communication under stringent resource constraints, addressing the limitations of imperfect CSI, and managing quantized RIS phase shifts.</p>
<p>BackCom emerges as a promising solution for energy-efficient data transfer in IoT networks, yet it suffers from double-fading effects, limited communication range, and security vulnerabilities. Moreover, the dense connectivity envisioned for 6G networks amplifies the risks of eavesdropping, necessitating advanced security strategies that align with resource-efficient operation.</p>
<p>RIS offer a paradigm shift by transforming wireless environments with their ability to manipulate electromagnetic waves. While traditional research treats RIS as a passive signal reflector, this work explores its potential as an active optimization component. By strategically configuring RIS, the limitations of backscatter communication can be mitigated, and both communication performance and security can be enhanced. Furthermore, the integration of advanced reinforcement learning techniques, such as the DDPG algorithm, provides a pathway for real-time adaptability in dynamic wireless environments.</p>
<p>This study offers the following distinctive contributions:
<list list-type="order">
<list-item>
<p><bold>Novel RIS-Enabled Joint Backscatter and Communication (JBAC) System with Imperfect CSI Management:</bold> We present a groundbreaking framework that seamlessly integrates RIS into IoT networks, revolutionizing both backscatter and direct communication modes. Unlike traditional approaches, our model tackles the challenge of imperfect CSI through an adaptive, learning-based strategy, ensuring robust system performance even in the face of channel estimation inaccuracies. By reimagining RIS as an active optimization component rather than a passive reflector, the framework dynamically optimizes signal paths, unlocking new levels of efficiency and reliability for next-generation IoT communication.</p></list-item>
<list-item>
<p><bold>Integration of PLS with Quantized RIS Phase Shifts:</bold> While existing PLS frameworks rely on idealized RIS setups, our study incorporates quantized RIS phase shifts a practical constraint in hardware implementations. We demonstrate how RIS selectively enhances signals toward legitimate users while degrading signals directed toward eavesdroppers, effectively mitigating eavesdropping risks even under quantization constraints.</p></list-item>
<list-item>
<p><bold>Dynamic Optimization via DDPG Algorithm in Continuous Action Spaces:</bold> To overcome challenges posed by continuous and high-dimensional action spaces, this work leverages the DDPG algorithm for real-time optimization of RIS configurations and power distribution. This approach not only adapts to environmental dynamics but also ensures the joint optimization of beamforming and RIS phase shifts, outperforming conventional optimization techniques.</p></list-item>
<list-item>
<p><bold>Addressing Backscatter-Specific Limitations: Double-Fading Effects and Coverage Range:</bold> Our innovative approach revolutionizes backscatter communication by overcoming critical challenges like double fading effects and limited coverage. By seamlessly integrating RIS-assisted signal propagation our framework unlocks new possibilities leveraging RIS to establish additional propagation paths. This not only enhances spectral efficiency but also effectively mitigates fading delivering robust performance in dense IoT environments and paving the way for next-generation connectivity solutions</p></list-item>
<list-item>
<p><bold>Benchmarking for Secure 6G Communication Protocols:</bold> By combining RIS functionality with advanced algorithmic methods and addressing practical constraints like imperfect CSI and quantized RIS phases this study establishes new benchmarks for secure and efficient 6G communication protocols. The results provide actionable insights for designing future ready IoT networks that prioritize both operational efficiency and data security.</p></list-item>
</list></p>
</sec>
</sec>
<sec id="s2">
<label>2</label>
<title>System Model and Problem Formulation</title>
<sec id="s2_1">
<label>2.1</label>
<title>System Model</title>
<p>We propose an integrated framework that utilizes BackCom with PLS in an RIS-aided environment as shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>. This model incorporates elements from both the bistatic BackCom system and the RIS-assisted wiretap communication system, providing a comprehensive approach to secure communication.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>System model of RIS-aided IoT network for 6G communication, enhancing both backscatter and direct modes</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_61744-fig-1.tif"/>
</fig>
<p>The system involves several components. A base station (BS) equipped along <italic>M</italic> antennas functions as both a Carrier Emitter (CE) for BackCom and a transmitter for direct communication. The setup involves two single-antenna devices: a Backscatter Device (BD) and a legitimate user referred to as Bob. An Eavesdropper (Eve), equipped with <italic>M</italic><sup><italic>&#x2032;</italic></sup> antennas, aims to intercept communications. Additionally, an RIS with <italic>N</italic> reflecting elements is employed to enhance signal propagation through adjustable phase shifts.</p>
<p>In this system, the Carrier Emitter (CE) broadcasts a continuous wave signal <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> with power <italic>P</italic> and a beamforming vector <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>. The power of the transmitted signal is subject to the constraint:
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:math></inline-formula> represents the beamforming vector.</p>
<p>The Base Station (BS) also performs direct communication with Bob by transmitting a signal <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mi>s</mml:mi></mml:math></inline-formula>, which satisfies the power condition <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mo fence="false" stretchy="false">&#x007C;</mml:mo><mml:mi>s</mml:mi><mml:msup><mml:mo fence="false" stretchy="false">&#x007C;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>. The beamforming vector <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:math></inline-formula> is constrained by:
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mrow><mml:mtext>max</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>The communication channels between different components are represented as follows: the link from the CE/BS to the RIS is denoted as <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>L</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> and <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>; the channel from the CE/BS to Bob and Eve are denoted as <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> and <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mi>M</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:mrow></mml:msup></mml:math></inline-formula>, respectively. Similarly, the RIS-to-receiver channels are represented by <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> for Bob and <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mi>M</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:mrow></mml:msup></mml:math></inline-formula> for Eve. Additionally, the direct channel from the Backscatter Device (BD) to Bob is given by <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow></mml:math></inline-formula>.</p>
<p>The RIS reflects incoming signals by applying phase shifts. These phase shifts are described by the matrix:
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mtext>diag</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mi>N</mml:mi></mml:msub></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> represents the phase shift of the <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:mi>i</mml:mi></mml:math></inline-formula>th reflecting element. These phase shifts influence both the backscatter communication and direct transmission, adding a degree of control to the signal propagation.</p>
<p>The received signal at Bob incorporates contributions from both direct communication and backscatter communication. It is given by:
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:msub><mml:mi>y</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>B</mml:mi></mml:mrow><mml:mi>H</mml:mi></mml:msubsup><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msubsup><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>B</mml:mi></mml:mrow><mml:mi>H</mml:mi></mml:msubsup><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>n</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:msub><mml:mi>n</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:mi>&#x1D49E;</mml:mi><mml:mi>&#x1D4A9;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mi>&#x03C3;</mml:mi><mml:mi>B</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is the additive noise at Bob. Similarly, the signal received by Eve is expressed as:
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:msub><mml:mi>y</mml:mi><mml:mi>E</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mi>H</mml:mi></mml:msubsup><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msubsup><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mi>H</mml:mi></mml:msubsup><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>n</mml:mi><mml:mi>E</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:msub><mml:mi>n</mml:mi><mml:mi>E</mml:mi></mml:msub><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:mi>&#x1D49E;</mml:mi><mml:mi>&#x1D4A9;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mi>&#x03C3;</mml:mi><mml:mi>E</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mrow><mml:mtext mathvariant="bold">I</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represents the additive noise at Eve.</p>
<p>The RIS enhances communication quality and secrecy by intelligently adjusting the phase shifts to strengthen the desired signals toward the legitimate user and reduce the signals leaking to the eavesdropper. This dual capability ensures efficient communication while mitigating the risk of eavesdropping.</p>
<p><italic>CSI Acquisition in the Proposed System</italic></p>
<p>Acquiring accurate CSI is essential for optimizing RIS configurations and ensuring secure communication. However, achieving perfect CSI in backscatter communication systems presents significant challenges due to double-fading effects, weaker received signals, and environmental dynamics. To address these issues, we adopt the following CSI acquisition strategies:
<list list-type="order">
<list-item>
<p><bold>Direct Links:</bold> CSI for direct links (e.g., BS-to-Bob and BS-to-Eve) is obtained through traditional pilot-based channel estimation. The BS transmits known pilot sequences to Bob, who estimates the channel response and provides feedback. For Eve, the system assumes partial CSI knowledge via statistical models or worst-case approximations to ensure robust optimization.</p></list-item>
<list-item>
<p><bold>Backscatter Links:</bold> - <italic>Active Probing:</italic> The BS sends probing signals that the BD modulates. The received composite signal at the BS is used to estimate the backscatter channel through advanced signal processing techniques such as sparse recovery. - <italic>RIS-Assisted Refinement:</italic> RIS configurations are adjusted iteratively to enhance the backscatter signal during estimation, enabling more accurate CSI acquisition.</p></list-item>
<list-item>
<p><bold>RIS-Specific Channels:</bold> - <italic>Cascaded Channel Estimation:</italic> The BS transmits pilots while the RIS adjusts its phase shifts sequentially. The BS estimates the combined BS-to-RIS and RIS-to-receiver channels using a cascaded model:
<disp-formula id="ueqn-37"><mml:math id="mml-ueqn-37" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mover><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mtext>cascade</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>R</mml:mi><mml:mi>I</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mi>I</mml:mi><mml:mi>S</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>U</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mrow><mml:mtext mathvariant="bold">E</mml:mtext></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
where <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mrow><mml:mtext mathvariant="bold">E</mml:mtext></mml:mrow></mml:math></inline-formula> captures the estimation error. - <italic>Decomposition via Sparse Recovery:</italic> Techniques like compressive sensing separate the individual components of the cascaded channel.</p></list-item>
<list-item>
<p><bold>Handling CSI Imperfections:</bold> To account for imperfect CSI, the system incorporates the following strategies:
<list list-type="bullet">
<list-item>
<p>Robust optimization techniques consider probabilistic CSI error bounds, optimizing beamforming vectors and RIS phase shifts for worst-case scenarios. The error model is represented as:
<disp-formula id="ueqn-38"><mml:math id="mml-ueqn-38" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mover><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext mathvariant="bold">E</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">E</mml:mtext></mml:mrow><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mo>&#x2264;</mml:mo><mml:mi>&#x03F5;</mml:mi><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
</list-item>
<list-item>
<p>The DDPG algorithm dynamically adapts to CSI imperfections, leveraging real-time learning to adjust configurations based on observed system performance.</p></list-item>
<list-item>
<p>A trade-off is maintained between the pilot overhead for CSI acquisition and system performance, ensuring a balance between accuracy and efficiency.</p></list-item>
</list></p></list-item>
</list></p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Problem Formulation</title>
<p>The optimization problem for the RIS-aided 6G communication system seeks to maximize the secrecy rate <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:msub><mml:mi>R</mml:mi><mml:mi>S</mml:mi></mml:msub></mml:math></inline-formula>, defined as the difference between the achievable rates at Bob (legitimate user) and Eve (eavesdropper):
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:msub><mml:mi>R</mml:mi><mml:mi>S</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mi>E</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>]</mml:mo></mml:mrow><mml:mo>+</mml:mo></mml:msup><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>The goal is to increase <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:msub><mml:mi>R</mml:mi><mml:mi>S</mml:mi></mml:msub></mml:math></inline-formula> by jointly enhancing the beamforming vector <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:math></inline-formula> and RIS phase shift matrix <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:mi mathvariant="bold">&#x03A6;</mml:mi></mml:math></inline-formula>:
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:munder><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi></mml:mrow></mml:munder><mml:mspace width="1em" /><mml:msub><mml:mi>R</mml:mi><mml:mi>S</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>subject to the following practical constraints:</p>
<p><bold>Constraints 1</bold>. <bold>Transmit Power Constraint:</bold> Overall power of the beamforming vector must not exceed the set maximum limit:
<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mrow><mml:mtext>max</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p><bold>2. RIS Phase Shift Constraints:</bold> Each RIS element&#x2019;s phase shift must be within the feasible range:
<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>and for quantized phase shifts:
<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi></mml:mrow><mml:mi>Q</mml:mi></mml:mfrac><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p><bold>3. Imperfect CSI:</bold> Robustness to CSI estimation errors is incorporated as:
<disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:mo movablelimits="true" form="prefix">Pr</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mover><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mo>&#x2264;</mml:mo><mml:mi>&#x03F5;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2265;</mml:mo><mml:mi>&#x03B4;</mml:mi><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:mrow><mml:mover><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow></mml:math></inline-formula> is the estimated channel, <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:mi>&#x03F5;</mml:mi></mml:math></inline-formula> the error tolerance, and <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:mi>&#x03B4;</mml:mi></mml:math></inline-formula> the confidence level.</p>
<p><bold>4. Energy Efficiency Constraint:</bold> The system ensures minimum energy efficiency:
<disp-formula id="eqn-12"><label>(12)</label><mml:math id="mml-eqn-12" display="block"><mml:mi>&#x03B7;</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>R</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mfrac><mml:mo>&#x2265;</mml:mo><mml:msub><mml:mi>&#x03B7;</mml:mi><mml:mrow><mml:mrow><mml:mtext>min</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:msub><mml:mi>&#x03B7;</mml:mi><mml:mrow><mml:mtext>min</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> is the threshold.</p>
<p><bold>5. Multi-Objective Trade-Off:</bold> A balance between secrecy and communication quality is captured as:
<disp-formula id="eqn-13"><label>(13)</label><mml:math id="mml-eqn-13" display="block"><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:msub><mml:mi>R</mml:mi><mml:mi>S</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:msub><mml:mi>R</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:math></inline-formula> control the relative importance of secrecy and quality.</p>
<p><bold>Problem Statement:</bold> Incorporating these constraints, the extended optimization problem is expressed as:
<disp-formula id="eqn-14"><label>(14)</label><mml:math id="mml-eqn-14" display="block"><mml:munder><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi></mml:mrow></mml:munder><mml:mspace width="1em" /><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mi>E</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>]</mml:mo></mml:mrow><mml:mo>+</mml:mo></mml:msup><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>subject to the transmit power constraint <xref ref-type="disp-formula" rid="eqn-8">(8)</xref>, RIS phase shift constraints <xref ref-type="disp-formula" rid="eqn-9">(9)</xref> and <xref ref-type="disp-formula" rid="eqn-10">(10)</xref>, imperfect CSI robustness <xref ref-type="disp-formula" rid="eqn-11">(11)</xref>, and energy efficiency requirements <xref ref-type="disp-formula" rid="eqn-12">(12)</xref>.</p>
<p>This optimization framework integrates practical considerations, ensuring efficient and secure RIS-aided operation under real-world conditions while addressing performance and hardware limitations.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>DRL Based Methodology for Secure Communication</title>
<p>DRL combines principles from reinforcement learning and deep learning to tackle complex decision-making challenges. By exploiting the analytical potential of deep neural networks, DRL empowers agents to navigate and adapt within intricate and dynamic environments, enabling effective decision-making based on vast and diverse datasets. This approach offers a robust framework for addressing tasks requiring adaptive and data-driven solutions in secure communication systems [<xref ref-type="bibr" rid="ref-35">35</xref>].</p>
<p>The foundation of DRL lies in training agents to determine optimal decisions by learning from iterative process experiences. Actors interact with their surroundings by observing its state, executing actions and receiving responses through rewards which guide their learning process. The goal is to formulate a policy that ensures the maximum total reward accumulated over time [<xref ref-type="bibr" rid="ref-36">36</xref>]. What distinguishes DRL from traditional reinforcement learning approaches is its utilization of deep learning. This integration enables the direct processing of raw sensory inputs, such as visual content or sensory information, excluding relying on physically engineered attributes. Deep neural networks act as powerful estimators, either for policies that identify optimal actions in given situations or for value functions that estimate expected rewards derived from current states and actions [<xref ref-type="bibr" rid="ref-37">37</xref>]. DRL offers notable benefits: Well-suited for managing intricate environments with extensive, high-dimensional state spaces, it is ideal for real-world use cases.</p>
<p>Additionally, agents using DRL leverage the generalization ability of neural networks to apply learned knowledge effectively in novel and unseen scenarios.</p>
<p>The capability for end-to-end learning from raw data to actions simplifies the overall learning process.</p>
<p>It has ability to adapt and scale according to problem size and complexity enables its application across a wide range of tasks, including mastering complex games and controlling autonomous vehicles.</p>
<p>Its ability to adapt and scale according to problem size and complexity enables its application across a wide range of tasks, including mastering complex games and controlling autonomous vehicles.</p>
<p>Nonetheless, challenges exist within DRL. DRL often requires substantial amounts of training data and can be susceptible to overfitting. It may also encounter challenges related to stability and reliability during the learning process, particularly in scenarios where rewards are rare or come with a delay. Despite these challenges, DRL marks a transformative advancement in artificial intelligence, providing a robust toolkit for addressing diverse decision-making problems in complex and ever-changing scenarios.</p>
<sec id="s3_1">
<label>3.1</label>
<title>MDP</title>
<p>A Markov Decision Process (MDP) serves as a mathematical framework to model decision-making in situations where outcomes are influenced by both the decision-maker&#x2019;s actions and elements of randomness. The key elements of an MDP, along with their mathematical formulations, are described as follows:
<list list-type="bullet">
<list-item>
<p><bold>State Space (<italic>S</italic>):</bold> The state space <italic>S</italic> defines the collection of all possible states within the environment. Each state <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mi>s</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>S</mml:mi></mml:math></inline-formula> represents the condition or configuration of the environment at a specific moment in time.</p></list-item>
<list-item>
<p><bold>Action Set (<italic>A</italic>):</bold> The action set <italic>A</italic> includes all the actions that the agent is able to take. For a particular state <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:mi>s</mml:mi></mml:math></inline-formula> the available actions are denoted as <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mi>a</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>A</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>. The action set may vary depending on the state as indicated by the notation <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mi>A</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>.</p></list-item>
<list-item>
<p><bold>Transition Dynamics (<italic>P</italic>):</bold> The transition dynamics are characterized by the probability function <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>s</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo>&#x2223;</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, which specifies the likelihood of transitioning to state <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:msup><mml:mi>s</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:math></inline-formula> from state <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:mi>s</mml:mi></mml:math></inline-formula> after performing action <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:mi>a</mml:mi></mml:math></inline-formula>. This function captures both deterministic and stochastic elements of the system dynamics:
<disp-formula id="eqn-15"><label>(15)</label><mml:math id="mml-eqn-15" display="block"><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>s</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo>&#x2223;</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo movablelimits="true" form="prefix">Pr</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mi>s</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>
where <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:msup><mml:mi>s</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:math></inline-formula> represents the state at time <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>, <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:mi>s</mml:mi></mml:math></inline-formula> is the state at time <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:mi>t</mml:mi></mml:math></inline-formula>, and <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:mi>a</mml:mi></mml:math></inline-formula> is the action taken.</p></list-item>
<list-item>
<p><bold>Reward Function (<italic>R</italic>):</bold> The reward function <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:mi>R</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mi>s</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> or <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:mi>R</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> defines the immediate reward obtained after transitioning from state <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:mi>s</mml:mi></mml:math></inline-formula> to state <inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:msup><mml:mi>s</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:math></inline-formula> as a result of action <inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mi>a</mml:mi></mml:math></inline-formula>. It reflects the benefit or cost linked to the transition:
<disp-formula id="eqn-16"><label>(16)</label><mml:math id="mml-eqn-16" display="block"><mml:mi>R</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mi>s</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mi>s</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo stretchy="false">]</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>
where <inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is the reward received at time <inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>.</p></list-item>
<list-item>
<p><bold>Discount Factor (<inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:mi>&#x03B3;</mml:mi></mml:math></inline-formula>):</bold> The discount factor <inline-formula id="ieqn-50"><mml:math id="mml-ieqn-50"><mml:mi>&#x03B3;</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> determines the weight given to future rewards. A higher value of <inline-formula id="ieqn-51"><mml:math id="mml-ieqn-51"><mml:mi>&#x03B3;</mml:mi></mml:math></inline-formula> (close to 1) means the agent prioritizes long-term rewards, while a lower value favors immediate rewards. The discount factor ensures convergence of the total reward and accounts for the uncertainty in future outcomes.</p></list-item>
</list></p>
<p>MDP is formally represented by <inline-formula id="ieqn-52"><mml:math id="mml-ieqn-52"><mml:mo stretchy="false">(</mml:mo><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>P</mml:mi><mml:mo>,</mml:mo><mml:mi>R</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>. The agent goal is to determine an optimal policy <inline-formula id="ieqn-53"><mml:math id="mml-ieqn-53"><mml:msup><mml:mi>&#x03C0;</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup></mml:math></inline-formula> that maximizes the expected sum of discounted future rewards. A policy <inline-formula id="ieqn-54"><mml:math id="mml-ieqn-54"><mml:mi>&#x03C0;</mml:mi></mml:math></inline-formula> maps states to actions and is denoted as:
<disp-formula id="ueqn-19"><mml:math id="mml-ueqn-19" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo movablelimits="true" form="prefix">Pr</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>The objective is to find the optimal policy <inline-formula id="ieqn-55"><mml:math id="mml-ieqn-55"><mml:msup><mml:mi>&#x03C0;</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup></mml:math></inline-formula> that maximizes the anticipated cumulative reward <inline-formula id="ieqn-56"><mml:math id="mml-ieqn-56"><mml:msub><mml:mi>G</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> beginning from a given state:
<disp-formula id="eqn-17"><label>(17)</label><mml:math id="mml-eqn-17" display="block"><mml:msup><mml:mi>&#x03C0;</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:mi>arg</mml:mi><mml:mo>&#x2061;</mml:mo><mml:munder><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mi>&#x03C0;</mml:mi></mml:munder><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mi>G</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>where the total discounted reward <inline-formula id="ieqn-57"><mml:math id="mml-ieqn-57"><mml:msub><mml:mi>G</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> from time <inline-formula id="ieqn-58"><mml:math id="mml-ieqn-58"><mml:mi>t</mml:mi></mml:math></inline-formula> onward is defined as:
<disp-formula id="eqn-18"><label>(18)</label><mml:math id="mml-eqn-18" display="block"><mml:msub><mml:mi>G</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x221E;</mml:mi></mml:mrow></mml:munderover><mml:msup><mml:mi>&#x03B3;</mml:mi><mml:mi>k</mml:mi></mml:msup><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>Solving an MDP involves identifying the optimal policy <inline-formula id="ieqn-59"><mml:math id="mml-ieqn-59"><mml:msup><mml:mi>&#x03C0;</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup></mml:math></inline-formula> that optimizes the expected cumulative rewards. Common approaches for solving MDPs, such as Q-Learning, Policy Iteration and Value Iteration, work by iteratively improving the policy or updating state and action values using the MDP&#x2019;s defined components.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Proposed DDPG Model</title>
<p>We apply the Markov Decision Process (MDP) framework to tackle the optimization hurdles in ensuring secure wireless communication systems, including both direct and backscatter communication. MDPs are well-suited for scenarios where current decisions&#x2013;such as selecting the beamforming vector and RIS phase shifts&#x2013;impact future states and rewards. This approach considers long-term consequences for the system&#x2019;s performance and secrecy, unlike straightforward optimization approaches that prioritize immediate results.</p>
<p>To ensure scalability in large-scale RIS deployments, we utilize DDPG, a RL-based framework that excels in managing large action spaces, such as those present in RIS systems with a high number of reflecting elements. DDPG efficiently adapts the system to dynamically fine-tune the phase shift of RIS and beamforming vectors, balancing the trade-off between communication quality and secrecy. It also leverages actor-critic networks, where the actor determines the optimal actions based on the current state, while the critic evaluates these actions by estimating their Q-values, representing the expected rewards.</p>
<p>This reinforcement learning-based approach enables the system to learn and adjust configurations in real-time, making it computationally feasible for large-scale environments, especially an increase in the number of RIS elements. Moreover, parallelization and distributed learning techniques can further enhance the scalability and computational efficiency, allowing the system to handle complex, large-scale RIS deployments in real-world scenarios.</p>
<p><bold>Actor Network: </bold>In our secure communication framework, the actor network defines the policy by linking the current environmental state <inline-formula id="ieqn-60"><mml:math id="mml-ieqn-60"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> to a group of actions <inline-formula id="ieqn-61"><mml:math id="mml-ieqn-61"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>, which includes the RIS phase shifts and beamforming vector elements. The policy is given as:
<disp-formula id="eqn-19"><label>(19)</label><mml:math id="mml-eqn-19" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-62"><mml:math id="mml-ieqn-62"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> is the action vector at time <inline-formula id="ieqn-63"><mml:math id="mml-ieqn-63"><mml:mi>t</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-64"><mml:math id="mml-ieqn-64"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> represents the current environmental state information along with the current vector for beamforming and <inline-formula id="ieqn-65"><mml:math id="mml-ieqn-65"><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup></mml:math></inline-formula> represents the actor network variables.</p>
<p>The action vector <inline-formula id="ieqn-66"><mml:math id="mml-ieqn-66"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> is structured such that the first <italic>N</italic> elements correspond to the RIS phase shifts <inline-formula id="ieqn-67"><mml:math id="mml-ieqn-67"><mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mtext>diag</mml:mtext><mml:mspace width="thinmathspace" /><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mi>N</mml:mi></mml:msub></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>. These Phase shifts are normalized to fall within the specified range. <inline-formula id="ieqn-68"><mml:math id="mml-ieqn-68"><mml:mo stretchy="false">[</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> to ensure valid values:
<disp-formula id="eqn-20"><label>(20)</label><mml:math id="mml-eqn-20" display="block"><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>The remaining elements of <inline-formula id="ieqn-69"><mml:math id="mml-ieqn-69"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> correspond actual and virtual elements associated with the vector components of beamforming <inline-formula id="ieqn-70"><mml:math id="mml-ieqn-70"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>, which must satisfy the transmit power constraint:
<disp-formula id="eqn-21"><label>(21)</label><mml:math id="mml-eqn-21" display="block"><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mrow><mml:mtext>max</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>To ensure compliance, the beamforming vector is normalized as:
<disp-formula id="eqn-22"><label>(22)</label><mml:math id="mml-eqn-22" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">&#x2190;</mml:mo><mml:mfrac><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msqrt><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mrow><mml:mtext>max</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:msqrt><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>By leveraging this approach, the agent can flexibly fine-tune the RIS phase adjustments and beamforming configuration according to the environment, optimizing transmission efficiency and ensuring secure communication.</p>
<p><bold>Critic Network: </bold>The role of the critic network is to evaluate the actions generated by the actor network by estimating their Q-values. A Q-value signifies the expected cumulative reward for executing an action <inline-formula id="ieqn-71"><mml:math id="mml-ieqn-71"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> in the state <inline-formula id="ieqn-72"><mml:math id="mml-ieqn-72"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> and continuing to follow the policy <inline-formula id="ieqn-73"><mml:math id="mml-ieqn-73"><mml:mi>&#x03C0;</mml:mi></mml:math></inline-formula>. The Q-value function is given as:
<disp-formula id="eqn-23"><label>(23)</label><mml:math id="mml-eqn-23" display="block"><mml:mi>Q</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>Q</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mi>Q</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:msup><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mspace width="thinmathspace" /><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.623em" minsize="1.623em">|</mml:mo></mml:mrow></mml:mstyle><mml:mspace width="thinmathspace" /><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>]</mml:mo></mml:mrow><mml:mspace width="negativethinmathspace" /><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-74"><mml:math id="mml-ieqn-74"><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>Q</mml:mi></mml:msup></mml:math></inline-formula> defines the parameters of the critic network, while <inline-formula id="ieqn-75"><mml:math id="mml-ieqn-75"><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:msup><mml:mi>Q</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:mrow></mml:msup></mml:math></inline-formula> represents the targeted critic network characteristics. The target critic network improves training stability by providing a consistent reference for updating Q-values.</p>
<p>The state <inline-formula id="ieqn-76"><mml:math id="mml-ieqn-76"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> consists of the CSI <inline-formula id="ieqn-77"><mml:math id="mml-ieqn-77"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, along with the current beamforming vector <inline-formula id="ieqn-78"><mml:math id="mml-ieqn-78"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> and RIS phase shifts <inline-formula id="ieqn-79"><mml:math id="mml-ieqn-79"><mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>. The action <inline-formula id="ieqn-80"><mml:math id="mml-ieqn-80"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> encompasses the phase adjustments of RIS and beamforming vector elements.</p>
<p><bold>Reward Function: </bold>The immediate reward at time <inline-formula id="ieqn-81"><mml:math id="mml-ieqn-81"><mml:mi>t</mml:mi></mml:math></inline-formula>, denoted by <inline-formula id="ieqn-82"><mml:math id="mml-ieqn-82"><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>, indicates the system, s secrecy performance, RIS phase stability, and energy efficiency. It is defined as:
<disp-formula id="eqn-24"><label>(24)</label><mml:math id="mml-eqn-24" display="block"><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x03B1;</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mi>E</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>]</mml:mo></mml:mrow><mml:mo>+</mml:mo></mml:msup><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:msubsup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mrow><mml:mtext>variation</mml:mtext></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>R</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mrow><mml:mtext>used</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mfrac><mml:mo>,</mml:mo></mml:math></disp-formula>where
<list list-type="bullet">
<list-item>
<p><inline-formula id="ieqn-83"><mml:math id="mml-ieqn-83"><mml:msub><mml:mi>R</mml:mi><mml:mi>B</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> and <inline-formula id="ieqn-84"><mml:math id="mml-ieqn-84"><mml:msub><mml:mi>R</mml:mi><mml:mi>E</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> are the achievable rates at Bob (legitimate user) and Eve (eavesdropper), respectively.</p></list-item>
<list-item>
<p><inline-formula id="ieqn-85"><mml:math id="mml-ieqn-85"><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mi>t</mml:mi><mml:msubsup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mtext>variation</mml:mtext></mml:mrow><mml:mn>2</mml:mn></mml:msubsup></mml:math></inline-formula> penalizes large variations in RIS phase shifts, ensuring stable configuration updates.</p></list-item>
<list-item>
<p><inline-formula id="ieqn-86"><mml:math id="mml-ieqn-86"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mtext>used</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula> denotes the total power used.</p></list-item>
<list-item>
<p><inline-formula id="ieqn-87"><mml:math id="mml-ieqn-87"><mml:mi>&#x03B1;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03B7;</mml:mi></mml:math></inline-formula> are weighting parameters that balance the trade-offs between secrecy, stability, and efficiency.</p></list-item>
</list></p>
<p><bold>Training Process: </bold>The DDPG algorithm iteratively trains the actor and critic networks using experience replay and target networks to ensure stable learning.</p>
<p>Transitions <inline-formula id="ieqn-88"><mml:math id="mml-ieqn-88"><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> are stored in a replay buffer <inline-formula id="ieqn-89"><mml:math id="mml-ieqn-89"><mml:mrow><mml:mi>&#x0212C;</mml:mi></mml:mrow></mml:math></inline-formula>. During training, a minibatch of transitions is sampled from <inline-formula id="ieqn-90"><mml:math id="mml-ieqn-90"><mml:mrow><mml:mi>&#x0212C;</mml:mi></mml:mrow></mml:math></inline-formula> to break correlations between consecutive steps and improve learning stability.</p>
<p>The critic network is optimized by minimizing the following loss function:
<disp-formula id="eqn-25"><label>(25)</label><mml:math id="mml-eqn-25" display="block"><mml:mi>L</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munder><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>Q</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>Q</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-91"><mml:math id="mml-ieqn-91"><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:msup><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msup><mml:mi>&#x03C0;</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:msup><mml:mi>&#x03C0;</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:msup><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represents the target Q-value, calculated with the help of the target networks.</p>
<p>The actor network is updated using the policy gradient:
<disp-formula id="eqn-26"><label>(26)</label><mml:math id="mml-eqn-26" display="block"><mml:msub><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup></mml:mrow></mml:msub><mml:mi>J</mml:mi><mml:mo>&#x2248;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munder><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msub><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mi>a</mml:mi></mml:msub><mml:mi>Q</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>Q</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.623em" minsize="1.623em">|</mml:mo></mml:mrow></mml:mstyle><mml:mrow><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup></mml:mrow></mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.623em" minsize="1.623em">|</mml:mo></mml:mrow></mml:mstyle><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>Target networks are updated slowly to ensure stability:
<disp-formula id="eqn-27"><label>(27)</label><mml:math id="mml-eqn-27" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:msup><mml:mi>&#x03C0;</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:mrow></mml:msup><mml:mo stretchy="false">&#x2190;</mml:mo><mml:mi>&#x03C4;</mml:mi><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03C4;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:msup><mml:mi>&#x03C0;</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-28"><label>(28)</label><mml:math id="mml-eqn-28" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:msup><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:mrow></mml:msup><mml:mo stretchy="false">&#x2190;</mml:mo><mml:mi>&#x03C4;</mml:mi><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>Q</mml:mi></mml:msup><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03C4;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:msup><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-92"><mml:math id="mml-ieqn-92"><mml:mi>&#x03C4;</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is the soft update coefficient.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Computational Complexity Analysis</title>
<p>The computational cost of the DDPG algorithm in RIS optimization is dictated by the architecture of the actor and critic networks, the scale of the state and action spaces (which are proportional to the number of RIS elements <italic>N</italic>), and the minibatch size <italic>B</italic> during the training process.</p>
<p>The actor network maps the current state <inline-formula id="ieqn-93"><mml:math id="mml-ieqn-93"><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>, which includes the channel state information (CSI) and RIS configuration states, to an action <inline-formula id="ieqn-94"><mml:math id="mml-ieqn-94"><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>, representing the adjustments in RIS phases (<italic>N</italic> elements) and vector components of beamforming. The critic network evaluates the Q-value of the selected action, providing feedback for policy improvement.</p>
<p>The total complexity for one training iteration can be expressed as:
<disp-formula id="eqn-29"><label>(29)</label><mml:math id="mml-eqn-29" display="block"><mml:mrow><mml:mi>&#x1D4AA;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>B</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:msup><mml:mi>h</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mrow><mml:mi>&#x1D4AA;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>B</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>N</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-95"><mml:math id="mml-ieqn-95"><mml:mi>l</mml:mi></mml:math></inline-formula> is the number of layers in the neural networks, <inline-formula id="ieqn-96"><mml:math id="mml-ieqn-96"><mml:mi>h</mml:mi></mml:math></inline-formula> is the number of neurons per layer, <italic>B</italic> is the minibatch size, and <italic>N</italic> is the number of RIS reflective elements.</p>
<p><bold>Scalability with <italic>N</italic>:</bold> The complexity grows linearly with <italic>N</italic> because the action space dimension is proportional to <italic>N</italic>. The phase shifts of each RIS element must be optimized, increasing the input size of the networks and the number of optimization variables. Despite this linear scaling, the use of parallelized computations in GPUs and efficient sampling in DDPG ensures scalability for practical deployments with RIS sizes of up to 200&#x2013;300 elements.</p>
<p><bold>Practical Considerations:</bold> Compact neural network designs with fewer layers and neurons (<inline-formula id="ieqn-97"><mml:math id="mml-ieqn-97"><mml:mi>l</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-98"><mml:math id="mml-ieqn-98"><mml:mi>h</mml:mi></mml:math></inline-formula>) reduce overhead while maintaining performance. Feature engineering or clustering techniques can also group RIS elements to simplify control in large-scale scenarios.</p>
<p>The scalability of the DDPG algorithm makes it suitable for RIS-aided 6G networks, even as the number of reflective elements increases.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Adaptive Learning for Enhanced Communication</title>
<p>To enhance security DDPG approach is used by the system for communication that iteratively updates the actor, critic networks influenced by the feedback from the environment, enabling continuous improvement of the policy.</p>
<p>The parameters of critic network&#x2019;s <inline-formula id="ieqn-99"><mml:math id="mml-ieqn-99"><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>Q</mml:mi></mml:msup></mml:math></inline-formula> improved by minimizing loss function <italic>L</italic>, which measures the error in the target Q-values and predicted Q-values computed from rewards and the target critic network&#x2019;s output:
<disp-formula id="eqn-30"><label>(30)</label><mml:math id="mml-eqn-30" display="block"><mml:mi>L</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munder><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mi>i</mml:mi></mml:munder><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>Q</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>Q</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>Here, <inline-formula id="ieqn-100"><mml:math id="mml-ieqn-100"><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> is the Q-value of target, defined as the reward plus the discounted Q-value of the subsequent state-action combination as estimated by the target critic network.</p>
<p>The actor network updates its parameters <inline-formula id="ieqn-101"><mml:math id="mml-ieqn-101"><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup></mml:math></inline-formula> to select actions that yield larger Q-values as evaluated by the network of critic. The actor&#x2019;s objective function <italic>J</italic> gradient is approximated through:
<disp-formula id="eqn-31"><label>(31)</label><mml:math id="mml-eqn-31" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup></mml:mrow></mml:msub><mml:mi>J</mml:mi></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>&#x2248;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munder><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msub><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mi>Q</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>Q</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="2.047em" minsize="2.047em">|</mml:mo></mml:mrow></mml:mstyle><mml:mrow><mml:mstyle scriptlevel="1"><mml:mtable rowspacing="0.1em" columnspacing="0em" displaystyle="false"><mml:mtr><mml:mtd><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mtext mathvariant="bold">a</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd /><mml:mtd><mml:mi></mml:mi><mml:mspace width="1em" /><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup></mml:mrow></mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="2.047em" minsize="2.047em">|</mml:mo></mml:mrow></mml:mstyle><mml:mrow><mml:msub><mml:mrow><mml:mtext mathvariant="bold">s</mml:mtext></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>This iterative training process aligns the actor&#x2019;s policy with the objective of maximizing rewards, improving both the efficiency and security of wireless communication.</p>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>State Space, Action</title>
<p>In the proposed framework for secure data transmission utilizing RIS-assisted backscatter technology, the Markov Decision Process (MDP) is modeled to optimize both transmission efficiency and security. The components of the MDP, namely the state space, action space, and reward mechanism, are described as follows:</p>
<p><bold>State Space:</bold> The state at time <inline-formula id="ieqn-102"><mml:math id="mml-ieqn-102"><mml:mi>t</mml:mi></mml:math></inline-formula>, denoted as <inline-formula id="ieqn-103"><mml:math id="mml-ieqn-103"><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>, encapsulates the real-time conditions of the environment, including channel state information (CSI) for all relevant communication links. It is formally represented as:
<disp-formula id="eqn-32"><label>(32)</label><mml:math id="mml-eqn-32" display="block"><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-104"><mml:math id="mml-ieqn-104"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>L</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> represents the CSI from the Base Station (BS) to the RIS, <inline-formula id="ieqn-105"><mml:math id="mml-ieqn-105"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> denotes the CSI from the transmitter to the RIS, and <inline-formula id="ieqn-106"><mml:math id="mml-ieqn-106"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> represents the CSI for the direct link between the BS and Bob (the legitimate user). The CSI for the RIS to Bob link is denoted by <inline-formula id="ieqn-107"><mml:math id="mml-ieqn-107"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, while <inline-formula id="ieqn-108"><mml:math id="mml-ieqn-108"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mi>M</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:mrow></mml:msup></mml:math></inline-formula> represents the RIS to Eve (eavesdropper) link. Additionally, <inline-formula id="ieqn-109"><mml:math id="mml-ieqn-109"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mi>M</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:mrow></mml:msup></mml:math></inline-formula> describes the direct channel from the BS to Eve, and <inline-formula id="ieqn-110"><mml:math id="mml-ieqn-110"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">h</mml:mtext></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi mathvariant="double-struck">C</mml:mi></mml:mrow></mml:math></inline-formula> captures the direct link from the Backscatter Device (BD) to Bob. These parameters collectively define the current state of the network, enabling informed decision-making for secure communication.</p>
<p><bold>Action Space:</bold> The action <inline-formula id="ieqn-111"><mml:math id="mml-ieqn-111"><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> is the decision variable at time <inline-formula id="ieqn-112"><mml:math id="mml-ieqn-112"><mml:mi>t</mml:mi></mml:math></inline-formula> and consists of two components: the beamforming vector <inline-formula id="ieqn-113"><mml:math id="mml-ieqn-113"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> and the RIS phase shift matrix <inline-formula id="ieqn-114"><mml:math id="mml-ieqn-114"><mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>. The action is determined as:</p>
<p><disp-formula id="eqn-33"><label>(33)</label><mml:math id="mml-eqn-33" display="block"><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="bold">&#x03A6;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D4A9;</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:math></disp-formula></p>
<p>where <inline-formula id="ieqn-115"><mml:math id="mml-ieqn-115"><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represents the actor network&#x2019;s policy function, mapping the current state from <inline-formula id="ieqn-116"><mml:math id="mml-ieqn-116"><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> to action <inline-formula id="ieqn-117"><mml:math id="mml-ieqn-117"><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-118"><mml:math id="mml-ieqn-118"><mml:msub><mml:mrow><mml:mi>&#x1D4A9;</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> is the introduced during training to promote diverse actions.</p>
<p>The action space is subject to the following constraints. The beamforming vector must satisfy the transmit power constraint:
<disp-formula id="eqn-34"><label>(34)</label><mml:math id="mml-eqn-34" display="block"><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mrow><mml:mtext>max</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-119"><mml:math id="mml-ieqn-119"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mtext>max</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> denotes the maximum allowable transmit power. The RIS phase shifts are constrained as:
<disp-formula id="eqn-35"><label>(35)</label><mml:math id="mml-eqn-35" display="block"><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-120"><mml:math id="mml-ieqn-120"><mml:msub><mml:mi>&#x03A8;</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> represents the phase shift of the <inline-formula id="ieqn-121"><mml:math id="mml-ieqn-121"><mml:mi>i</mml:mi></mml:math></inline-formula>-th RIS element. These constraints ensure that the actions taken are both physically feasible and compliant with the system&#x2019;s hardware limitations.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Results</title>
<p>In <xref ref-type="fig" rid="fig-2">Fig. 2</xref>, the &#x201C;Episode Rewards Over Time&#x201D; graph illustrates the progression of the DDPG algorithm in optimizing RIS configurations for secure communication in a 6G IoT network (Algorithm 1). Initially, during the exploration phase (0 to 250 episodes), rewards remain low as the algorithm evaluates diverse strategies. This is followed by a significant increase in rewards (250 to 750 episodes), signaling the identification and exploitation of effective actions for improving security. Finally, the rewards stabilize (750 to 2000 episodes), indicating policy convergence and consistent performance. The results highlight the DDPG algorithm, s ability to dynamically adapt and optimize RIS configurations, reinforcing its utility in enhancing secure communication within the 6G IoT environment.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Episode vs. rewards</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_61744-fig-2.tif"/>
</fig>
<fig id="fig-7">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_61744-fig-7.tif"/>
</fig>
<p>In <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, the &#x201C;Secrecy Rate vs. Transmit Power&#x201D; graph compares the proposed DDPG-based approach with the Alternate Optimization method and a system excluded RIS. The results demonstrate that the DDPG approach consistently delivers the high secrecy rate, with a steeper improvement as transmit power increases, highlighting its efficiency in optimizing RIS configurations for secure communication. In contrast, the AO method delivers moderate performance, while the system without RIS performs poorly due to its inability to enhance signal security. This demonstrates the DDPG approach&#x2019;s superiority in achieving higher secrecy rates, particularly at higher transmit powers, reinforcing its applicability for secure and efficient 6G IoT networks.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Secrecy rate vs. transmit power</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_61744-fig-3.tif"/>
</fig>
<p>In <xref ref-type="fig" rid="fig-4">Fig. 4</xref>, the impact of different learning rates on the DDPG algorithm, s performance in optimizing RIS configurations for secure communication is illustrated. A learning rate of 0.0001 achieves the highest average reward (17 to 18) with stable convergence, showcasing its effectiveness in balancing exploration and exploitation. Conversely, a learning rate of 0.001 stabilizes at a moderate reward level (12 to 14), while 0.00001 exhibits the lowest rewards (9 to 10) and higher variability, indicating slower and less reliable learning. These results emphasize the critical role of learning rate selection in enhancing the stability and efficiency of the DDPG algorithm for RIS-aided secure communication in 6G IoT networks. Further exploration of learning rates could refine these insights, enabling more robust algorithm design for secure communication systems.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Learning performance for different rates</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_61744-fig-4.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig-5">Fig. 5</xref> evaluates the scalability of the proposed DDPG algorithm by analyzing the secrecy rate dependent on the number of reflective elements of RIS. The results indicate a consistent increase in secrecy rate with more RIS elements, demonstrating the method, s effectiveness in optimizing larger RIS configurations. The DDPG algorithm outperforms the Alternate Optimization (AO) and Deep Q-Network (DQN) methods, showcasing superior handling of the expanded action space introduced by larger RIS arrays. As the number of elements increases, the DDPG framework dynamically adjusts RIS phase shifts and beamforming parameters, achieving higher secrecy rates and enhanced communication security. Notably, even with 120 RIS elements, the algorithm maintains computational efficiency, emphasizing its practicality for large-scale 6G IoT deployments. These findings highlight the DDPG algorithm&#x2019;s scalability and its suitability for real-world RIS-aided secure communication systems.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Secrecy rate vs. RIS elements over different techniques</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_61744-fig-5.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig-6">Fig. 6</xref> illustrates the critical role of RIS phase optimization in enhancing secrecy rates as the number of reflective elements increases. With optimized RIS phases the secrecy rate consistently improves, reaching approximately 7 bps/Hz for 100 elements. In contrast, non-optimized RIS phases show slower improvements achieving only around 3.5 bps/Hz. This stark contrast underscores the importance of phase optimization in maximizing system security and performance in RIS-aided networks. As the number of elements grows, phase optimization becomes indispensable for achieving secure and efficient communication, particularly in advanced 6G IoT applications.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>RIS phase optimization vs. non-optimized RIS</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_61744-fig-6.tif"/>
</fig>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusion</title>
<p>This study illuminates the transformative potential of RIS in revolutionizing 6G communication for IoT networks. By redefining participant R a passive IS as an active than reflector, we unlock pathways f new or c optimizing communication efficiency and spectral utilization rather ion in both backscatter and direct communication modes. A standout achievement of this research is the innovative integration of PLS with the DDPG algorithm, a powerful combination that enhances security and operational efficacy. Through meticulous strategic configurations of RIS and optimized power distribution, we achieve remarkable improvements in transmission efficiency while effectively addressing critical security challenges. Ultimately, this research sets pioneering benchmarks for the future of 6G communication protocols, laying a robust foundation for the creation of secure, interconnected, and highly efficient wireless systems. This study also addresses the computational complexity of the DDPG algorithm in the context of RIS optimization. By demonstrating that the algorithm&#x2019;s complexity scales linearly with the number of RIS elements, we establish its feasibility for real-world 6G deployments. The results validate that the performance gains outweigh the computational costs, ensuring scalability and efficiency for large-scale RIS-assisted networks. As we advance toward the next generation of networks, the methodologies and insights presented here will serve as a guiding light, shaping the future landscape of secure IoT communications.</p>
</sec>
</body>
<back>
<ack>
<p>Not applicable.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>The project was funded by the deanship of scientific research (DSR), King Abdukaziz University, Jeddah, under grant No. (G-1436-611&#x2013;225). The authors, therefore acknowledge with thanks DSR technical and financial support.</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>The authors declare their contributions to the paper as follows: study conception and design: Syed Zain Ul Abideen, Mian Muhammad Kamal; data collection: Mian Muhammad Kamal, Eaman Alharbi, Ashfaq Ahmad Malik; analysis and interpretation of results: Syed Zain Ul Abideen, Wadee Alhalabi, Muhammad Shahid Anwar; draft manuscript preparation: Mian Muhammad Kamal, Muhammad Shahid Anwar, Liaqat Ali. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>The data used in this study are available upon request, though some may be restricted due to privacy, confidentiality, or ethical concerns.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Towards smart and reconfigurable environment: intelligent reflecting surface aided wireless network</article-title>. <source>IEEE Commun Magaz</source>. <year>2019</year>;<volume>58</volume>(<issue>1</issue>):<fpage>106</fpage>&#x2013;<lpage>12</lpage>. doi:<pub-id pub-id-type="doi">10.1109/MCOM.001.1900107</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Sheen</surname> <given-names>B</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Feng</surname> <given-names>X</given-names></string-name>, <string-name><surname>Chowdhury</surname> <given-names>MMU</given-names></string-name></person-group>. <article-title>A deep learning based modeling of reconfigurable intelligent surface assisted wireless communications for phase shift configuration</article-title>. <source>IEEE Open J Commun Soc</source>. <year>2021</year>;<volume>2</volume>:<fpage>262</fpage>&#x2013;<lpage>72</lpage>. doi:<pub-id pub-id-type="doi">10.1109/OJCOMS.2021.3050119</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Jiang</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>W</given-names></string-name>, <string-name><surname>Peng</surname> <given-names>M</given-names></string-name>, <string-name><surname>Peng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Feng</surname> <given-names>M</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Backscatter communication meets practical battery-free Internet of Things: a survey and outlook</article-title>. <source>IEEE Commun Surv Tutor</source>. <year>2023</year>;<volume>25</volume>(<issue>3</issue>):<fpage>2021</fpage>&#x2013;<lpage>51</lpage>. doi:<pub-id pub-id-type="doi">10.1109/COMST.2023.3278239</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Sood</surname> <given-names>S</given-names></string-name></person-group>. <article-title>An overview of backscatter communication technique for performing wireless sensing in green communication networks</article-title>. In: <conf-name>2023 International Conference on Power Energy, Environment &#x0026; Intelligent Control (PEEIC)</conf-name>; <year>2023</year>; <publisher-loc>Greater Noida, India</publisher-loc>: <publisher-name>IEEE</publisher-name>. p. <fpage>7</fpage>&#x2013;<lpage>11</lpage>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tan</surname> <given-names>F</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>H</given-names></string-name>, <string-name><surname>Li</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Energy-efficient beamforming optimization for MISO communication based on reconfigurable intelligent surface</article-title>. <source>Phys Commun</source>. <year>2023</year>;<volume>57</volume>(<issue>3</issue>):<fpage>101996</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.phycom.2022.101996</pub-id>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Mu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Hou</surname> <given-names>T</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Di Renzo</surname> <given-names>M</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Reconfigurable intelligent surfaces: principles and opportunities</article-title>. <source>IEEE Commun Surv Tutor</source>. <year>2021</year>;<volume>23</volume>(<issue>3</issue>):<fpage>1546</fpage>&#x2013;<lpage>77</lpage>. doi:<pub-id pub-id-type="doi">10.1109/COMST.2021.3077737</pub-id>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Huang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Zappone</surname> <given-names>A</given-names></string-name>, <string-name><surname>Alexandropoulos</surname> <given-names>GC</given-names></string-name>, <string-name><surname>Debbah</surname> <given-names>M</given-names></string-name>, <string-name><surname>Yuen</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Reconfigurable intelligent surfaces for energy efficiency in wireless communication</article-title>. <source>IEEE Transact Wireless Commun</source>. <year>2019</year>;<volume>18</volume>(<issue>8</issue>):<fpage>4157</fpage>&#x2013;<lpage>70</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TWC.2019.2922609</pub-id>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>X</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Guan</surname> <given-names>YL</given-names></string-name>, <string-name><surname>Yuen</surname> <given-names>C</given-names></string-name>, <string-name><surname>Di Renzo</surname> <given-names>M</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Spectrum-learning-aided reconfigurable intelligent surfaces for &#x201C;green&#x201D; 6G networks</article-title>. <source>IEEE Network</source>. <year>2021</year>;<volume>35</volume>(<issue>6</issue>):<fpage>20</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1109/MNET.110.2100301</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ahmad</surname> <given-names>S</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>KS</given-names></string-name>, <string-name><surname>Naeem</surname> <given-names>F</given-names></string-name>, <string-name><surname>Tariq</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Resource allocation for IRS-assisted networks: a deep reinforcement learning approach</article-title>. <source>IEEE Commun Stand Magaz</source>. <year>2023</year>;<volume>7</volume>(<issue>3</issue>):<fpage>48</fpage>&#x2013;<lpage>55</lpage>. doi:<pub-id pub-id-type="doi">10.1109/MCOMSTD.0002.2200007</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zou</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Hanzo</surname> <given-names>L</given-names></string-name></person-group>. <article-title>A survey on wireless security: technical challenges, recent advances, and future trends</article-title>. <source>Proc IEEE</source>. <year>2016</year>;<volume>104</volume>(<issue>9</issue>):<fpage>1727</fpage>&#x2013;<lpage>65</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JPROC.2016.2558521</pub-id>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cui</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>G</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Secure wireless communication via intelligent reflecting surface</article-title>. <source>IEEE Wirel Commun Lett</source>. <year>2019</year>;<volume>8</volume>(<issue>5</issue>):<fpage>1410</fpage>&#x2013;<lpage>4</lpage>. doi:<pub-id pub-id-type="doi">10.1109/LWC.2019.2919685</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Naeem</surname> <given-names>F</given-names></string-name>, <string-name><surname>Ali</surname> <given-names>M</given-names></string-name>, <string-name><surname>Kaddoum</surname> <given-names>G</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Yuen</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Security and privacy for reconfigurable intelligent surface in 6G: a review of prospective applications and challenges</article-title>. <source>IEEE Open J Commun Soc</source>. <year>2023</year>;<volume>4</volume>:<fpage>1196</fpage>&#x2013;<lpage>217</lpage>. doi:<pub-id pub-id-type="doi">10.1109/OJCOMS.2023.3273507</pub-id>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Peng</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Kong</surname> <given-names>L</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>C</given-names></string-name>, <string-name><surname>Li</surname> <given-names>L</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Deep reinforcement learning for RIS-aided multiuser full-duplex secure communications with hardware impairments</article-title>. <source>IEEE Int Things J</source>. <year>2022</year>;<volume>9</volume>(<issue>21</issue>):<fpage>21121</fpage>&#x2013;<lpage>35</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JIOT.2022.3177705</pub-id>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gong</surname> <given-names>S</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Hoang</surname> <given-names>DT</given-names></string-name>, <string-name><surname>Niyato</surname> <given-names>D</given-names></string-name>, <string-name><surname>Shu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>DI</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Toward smart wireless communications via intelligent reflecting surfaces: a contemporary survey</article-title>. <source>IEEE Commun Surv Tutor</source>. <year>2020</year>;<volume>22</volume>(<issue>4</issue>):<fpage>2283</fpage>&#x2013;<lpage>314</lpage>. doi:<pub-id pub-id-type="doi">10.1109/COMST.2020.3004197</pub-id>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Pan</surname> <given-names>C</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>K</given-names></string-name>, <string-name><surname>Kolb</surname> <given-names>JF</given-names></string-name>, <string-name><surname>Elkashlan</surname> <given-names>M</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>M</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Reconfigurable intelligent surfaces for 6G systems: principles, applications, and research directions</article-title>. <source>IEEE Commun Magaz</source>. <year>2021</year>;<volume>59</volume>(<issue>6</issue>):<fpage>14</fpage>&#x2013;<lpage>20</lpage>. doi:<pub-id pub-id-type="doi">10.1109/MCOM.001.2001076</pub-id>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yuan</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>YJA</given-names></string-name>, <string-name><surname>Shi</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>W</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Reconfigurable-intelligent-surface empowered wireless communications: challenges and opportunities</article-title>. <source>IEEE Wireless Commun</source>. <year>2021</year>;<volume>28</volume>(<issue>2</issue>):<fpage>136</fpage>&#x2013;<lpage>43</lpage>. doi:<pub-id pub-id-type="doi">10.1109/MWC.001.2000256</pub-id>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Alhammadi</surname> <given-names>A</given-names></string-name>, <string-name><surname>Shayea</surname> <given-names>I</given-names></string-name>, <string-name><surname>El-Saleh</surname> <given-names>AA</given-names></string-name>, <string-name><surname>Azmi</surname> <given-names>MH</given-names></string-name>, <string-name><surname>Ismail</surname> <given-names>ZH</given-names></string-name>, <string-name><surname>Kouhalvandi</surname> <given-names>L</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Artificial intelligence in 6G wireless networks: opportunities, applications, and challenges</article-title>. <source>Int J Intell Syst</source>. <year>2024</year>;<volume>2024</volume>(<issue>1</issue>):<fpage>8845070</fpage>. doi:<pub-id pub-id-type="doi">10.1155/2024/8845070</pub-id>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Hong</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Intelligent reflecting surface aided communication systems: performance analysis</article-title>. In: <conf-name>2021 IEEE 32nd Annual International Symposium on Personal, Indoor and Mobile Radio Communications (PIMRC)</conf-name>; <year>2021</year>; <publisher-loc>Helsinki, Finland</publisher-loc>: <publisher-name>IEEE</publisher-name>. p. <fpage>519</fpage>&#x2013;<lpage>24</lpage>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kim</surname> <given-names>SH</given-names></string-name>, <string-name><surname>Park</surname> <given-names>SY</given-names></string-name>, <string-name><surname>Choi</surname> <given-names>KW</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>TJ</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>DI</given-names></string-name></person-group>. <article-title>Backscatter-aided cooperative transmission in wireless-powered heterogeneous networks</article-title>. <source>IEEE Transact Wireless Commun</source>. <year>2020</year>;<volume>19</volume>(<issue>11</issue>):<fpage>7309</fpage>&#x2013;<lpage>23</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TWC.2020.3010544</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liang</surname> <given-names>YC</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Long</surname> <given-names>R</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>H</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Backscatter communication assisted by reconfigurable intelligent surfaces</article-title>. <source>Proc IEEE</source>. <year>2022</year>;<volume>110</volume>(<issue>9</issue>):<fpage>1339</fpage>&#x2013;<lpage>57</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JPROC.2022.3169622</pub-id>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Zeng</surname> <given-names>M</given-names></string-name>, <string-name><surname>Bedeer</surname> <given-names>E</given-names></string-name>, <string-name><surname>Li</surname> <given-names>X</given-names></string-name>, <string-name><surname>Pham</surname> <given-names>QV</given-names></string-name>, <string-name><surname>Dobre</surname> <given-names>OA</given-names></string-name>, <string-name><surname>Fortier</surname> <given-names>P</given-names></string-name>, <etal>et al</etal></person-group>. <chapter-title>IRS-empowered wireless communications: state-of-the-art, key techniques, and open issues</chapter-title>. In: <source>6G wireless</source>. <publisher-loc>UK</publisher-loc>: <publisher-name>CRC Press</publisher-name>; <year>2021</year>. p. <fpage>15</fpage>&#x2013;<lpage>38</lpage>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>H</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Movable antenna-enabled RIS-aided integrated sensing and communication</article-title>. <comment>arXiv:240703228. 2024</comment>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Byun</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>H</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>S</given-names></string-name>, <string-name><surname>Shim</surname> <given-names>B</given-names></string-name></person-group>. <article-title>Channel estimation and phase shift control for UAV-carried RIS communication systems</article-title>. <source>IEEE Transact Vehic Technol</source>. <year>2023</year>;<volume>72</volume>(<issue>10</issue>):<fpage>13695</fpage>&#x2013;<lpage>700</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TVT.2023.3274871</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Niu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Chu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Xiao</surname> <given-names>P</given-names></string-name>, <string-name><surname>Nguyen</surname> <given-names>HX</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Joint beamforming design for secure RIS-assisted IoT networks</article-title>. <source>IEEE Int Things J</source>. <year>2022</year>;<volume>10</volume>(<issue>2</issue>):<fpage>1628</fpage>&#x2013;<lpage>41</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JIOT.2022.3210115</pub-id>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Do</surname> <given-names>DT</given-names></string-name>, <string-name><surname>Le</surname> <given-names>AT</given-names></string-name>, <string-name><surname>Ha</surname> <given-names>NDX</given-names></string-name>, <string-name><surname>Dao</surname> <given-names>NN</given-names></string-name></person-group>. <article-title>Physical layer security for Internet of Things via reconfigurable intelligent surface</article-title>. <source>Future Generat Comput Syst</source>. <year>2022</year>;<volume>126</volume>(<issue>1</issue>):<fpage>330</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.future.2021.08.012</pub-id>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Nguyen</surname> <given-names>KK</given-names></string-name>, <string-name><surname>Masaracchia</surname> <given-names>A</given-names></string-name>, <string-name><surname>Sharma</surname> <given-names>V</given-names></string-name>, <string-name><surname>Poor</surname> <given-names>HV</given-names></string-name>, <string-name><surname>Duong</surname> <given-names>TQ</given-names></string-name></person-group>. <article-title>RIS-assisted UAV communications for IoT with wireless power transfer using deep reinforcement learning</article-title>. <source>IEEE J Select Top Signal Process</source>. <year>2022</year>;<volume>16</volume>(<issue>5</issue>):<fpage>1086</fpage>&#x2013;<lpage>96</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JSTSP.2022.3172587</pub-id>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Faisal</surname> <given-names>K</given-names></string-name>, <string-name><surname>Choi</surname> <given-names>W</given-names></string-name></person-group>. <article-title>Machine learning approaches for reconfigurable intelligent surfaces: a survey</article-title>. <source>IEEE Access</source>. <year>2022</year>;<volume>10</volume>(<issue>10</issue>):<fpage>27343</fpage>&#x2013;<lpage>67</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ACCESS.2022.3157651</pub-id>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhou</surname> <given-names>H</given-names></string-name>, <string-name><surname>Erol-Kantarci</surname> <given-names>M</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Poor</surname> <given-names>HV</given-names></string-name></person-group>. <article-title>A survey on model-based, heuristic, and machine learning optimization approaches in RIS-aided wireless networks</article-title>. <source>IEEE Commun Surv Tutor</source>. <year>2024</year>;<volume>26</volume>(<issue>2</issue>):<fpage>781</fpage>&#x2013;<lpage>823</lpage>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Puspitasari</surname> <given-names>AA</given-names></string-name>, <string-name><surname>An</surname> <given-names>TT</given-names></string-name>, <string-name><surname>Alsharif</surname> <given-names>MH</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>BM</given-names></string-name></person-group>. <article-title>Emerging technologies for 6G communication networks: machine learning approaches</article-title>. <source>Sensors</source>. <year>2023</year>;<volume>23</volume>(<issue>18</issue>):<fpage>7709</fpage>. doi:<pub-id pub-id-type="doi">10.3390/s23187709</pub-id>; <pub-id pub-id-type="pmid">37765765</pub-id></mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Pogaku</surname> <given-names>AC</given-names></string-name>, <string-name><surname>Do</surname> <given-names>DT</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>BM</given-names></string-name>, <string-name><surname>Nguyen</surname> <given-names>ND</given-names></string-name></person-group>. <article-title>UAV-assisted RIS for future wireless communications: a survey on optimization and performance analysis</article-title>. <source>IEEE Access</source>. <year>2022</year>;<volume>10</volume>(<issue>4</issue>):<fpage>16320</fpage>&#x2013;<lpage>36</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ACCESS.2022.3149054</pub-id>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Basharat</surname> <given-names>S</given-names></string-name>, <string-name><surname>Hassan</surname> <given-names>SA</given-names></string-name>, <string-name><surname>Pervaiz</surname> <given-names>H</given-names></string-name>, <string-name><surname>Mahmood</surname> <given-names>A</given-names></string-name>, <string-name><surname>Ding</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Gidlund</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Reconfigurable intelligent surfaces: potentials, applications, and challenges for 6G wireless networks</article-title>. <source>IEEE Wireless Commun</source>. <year>2021</year>;<volume>28</volume>(<issue>6</issue>):<fpage>184</fpage>&#x2013;<lpage>91</lpage>. doi:<pub-id pub-id-type="doi">10.1109/MWC.011.2100016</pub-id>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Xiao</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xiong</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhuang</surname> <given-names>W</given-names></string-name></person-group>. <article-title>Learning-based reliable and secure transmission for UAV-RIS-assisted communication systems</article-title>. <source>IEEE Trans Wirel Commun</source>. <year>2024</year>;<volume>23</volume>(<issue>7</issue>):<fpage>6954</fpage>&#x2013;<lpage>67</lpage>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>K</given-names></string-name>, <string-name><surname>Li</surname> <given-names>X</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Tsiftsis</surname> <given-names>TA</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Deep reinforcement learning-based energy efficiency optimization for RIS-aided integrated satellite-aerial-terrestrial relay networks</article-title>. <source>IEEE Trans Commun</source>. <year>2024</year>;<volume>72</volume>(<issue>7</issue>):<fpage>4163</fpage>&#x2013;<lpage>78</lpage>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Guo</surname> <given-names>K</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Li</surname> <given-names>X</given-names></string-name>, <string-name><surname>Song</surname> <given-names>H</given-names></string-name>, <string-name><surname>Kumar</surname> <given-names>N</given-names></string-name></person-group>. <article-title>Deep reinforcement learning and NOMA-based multi-objective RIS-assisted IS-UAV-TNs: trajectory optimization and beamforming design</article-title>. <source>IEEE Transact Intell Transport Syst</source>. <year>2023</year>;<volume>24</volume>(<issue>9</issue>):<fpage>10197</fpage>&#x2013;<lpage>210</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TITS.2023.3267607</pub-id>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Fran&#x00E7;ois-Lavet</surname> <given-names>V</given-names></string-name>, <string-name><surname>Henderson</surname> <given-names>P</given-names></string-name>, <string-name><surname>Islam</surname> <given-names>R</given-names></string-name>, <string-name><surname>Bellemare</surname> <given-names>MG</given-names></string-name>, <string-name><surname>Pineau</surname> <given-names>J</given-names></string-name></person-group>. <article-title>An introduction to deep reinforcement learning</article-title>. <source>Foundat Trends&#x00AE; Mach Learn</source>. <year>2018</year>;<volume>11</volume>(<issue>3&#x2013;4</issue>):<fpage>219</fpage>&#x2013;<lpage>354</lpage>. doi:<pub-id pub-id-type="doi">10.1561/2200000071</pub-id>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Liang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>D</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Deep reinforcement learning: a survey</article-title>. <source>IEEE Transact Neural Netw Learn Syst</source>. <year>2022</year>;<volume>35</volume>(<issue>4</issue>):<fpage>5064</fpage>&#x2013;<lpage>78</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TNNLS.2022.3207346</pub-id>; <pub-id pub-id-type="pmid">36170386</pub-id></mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kaelbling</surname> <given-names>LP</given-names></string-name>, <string-name><surname>Littman</surname> <given-names>ML</given-names></string-name>, <string-name><surname>Moore</surname> <given-names>AW</given-names></string-name></person-group>. <article-title>Reinforcement learning: a survey</article-title>. <source>J Artif Intell Res</source>. <year>1996</year>;<volume>4</volume>:<fpage>237</fpage>&#x2013;<lpage>85</lpage>. doi:<pub-id pub-id-type="doi">10.1613/jair.301</pub-id>.</mixed-citation></ref>
</ref-list>
</back></article>