<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CSSE</journal-id>
<journal-id journal-id-type="nlm-ta">CSSE</journal-id>
<journal-id journal-id-type="publisher-id">CSSE</journal-id>
<journal-title-group>
<journal-title>Computer Systems Science &#x0026; Engineering</journal-title>
</journal-title-group>
<issn pub-type="ppub">0267-6192</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">34461</article-id>
<article-id pub-id-type="doi">10.32604/csse.2023.034461</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Double Deep Q-Network Method for Energy Efficiency and Throughput in a UAV-Assisted Terrestrial Network</article-title><alt-title alt-title-type="left-running-head">Double Deep Q-Network Method for Energy Efficiency and Throughput in a UAV-Assisted Terrestrial Network</alt-title><alt-title alt-title-type="right-running-head">Double Deep Q-Network Method for Energy Efficiency and Throughput in a UAV-Assisted Terrestrial Network</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Ouamri</surname><given-names>Mohamed Amine</given-names></name>
<xref ref-type="aff" rid="aff-1">1</xref>
<xref ref-type="aff" rid="aff-2">2</xref>
</contrib>
<contrib id="author-2" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Alkanhel</surname><given-names>Reem</given-names></name>
<xref ref-type="aff" rid="aff-3">3</xref><email>rialkanhal@pnu.edu.sa</email>
</contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Singh</surname><given-names>Daljeet</given-names></name>
<xref ref-type="aff" rid="aff-4">4</xref>
</contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>El-kenaway</surname><given-names>El-sayed M.</given-names></name>
<xref ref-type="aff" rid="aff-5">5</xref>
</contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Ghoneim</surname><given-names>Sherif S. M.</given-names></name>
<xref ref-type="aff" rid="aff-6">6</xref>
</contrib>
<aff id="aff-1"><label>1</label><institution>University Grenoble Alpes, CNRS, Grenoble INP, LIG, DRAKKAR Teams</institution>, <addr-line>38000, Grenoble</addr-line>, <country>France</country></aff>
<aff id="aff-2"><label>2</label><institution>Laboratoire d&#x2019;informatique M&#x00E9;dical, Universit&#x00E9; de Bejaia</institution>, <addr-line>Targa Ouzemour, Q22R&#x002B;475</addr-line>, <country>Algeria</country></aff>
<aff id="aff-3"><label>3</label><institution>Department of Information Technology, College of Computer and Information Sciences, Princess Nourah bint Abdulrahman University</institution>, <addr-line>P.O.Box 84428, Riyadh, 11671</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-4"><label>4</label><institution>Department of Research and Development, Centre for Space Research, School of Electronics and Electrical Engineering, Lovely Professional University</institution>, <addr-line>Phagwara, 144411</addr-line>, <country>India</country></aff>
<aff id="aff-5"><label>5</label><institution>Department of Communication and Electronics, Delta Higher Institute of Engineering and Technology</institution>, <addr-line>Mansoura</addr-line>, <country>Egypt</country></aff>
<aff id="aff-6"><label>6</label><institution>Electrical Engineering Department, College of Engineering, Taif University</institution>, <addr-line>P. O. BOX 11099, Taif, 21944</addr-line>, <country>Saudi Arabia</country></aff>
</contrib-group><author-notes><corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Reem Alkanhel. Email: <email>rialkanhal@pnu.edu.sa</email></corresp></author-notes>
<pub-date date-type="collection" publication-format="electronic"><year>2023</year></pub-date>
<pub-date date-type="pub" publication-format="electronic"><day>17</day><month>1</month><year>2023</year></pub-date>
<volume>46</volume>
<issue>1</issue>
<fpage>73</fpage>
<lpage>92</lpage>
<history>
<date date-type="received"><day>18</day><month>7</month><year>2022</year></date>
<date date-type="accepted"><day>22</day><month>9</month><year>2022</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2023 Ouamri et al.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Ouamri et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CSSE_34461.pdf"></self-uri>
<abstract><p>Increasing the coverage and capacity of cellular networks by deploying additional base stations is one of the fundamental objectives of fifth-generation (5G) networks. However, it leads to performance degradation and huge spectral consumption due to the massive densification of connected devices and simultaneous access demand. To meet these access conditions and improve Quality of Service, resource allocation (RA) should be carefully optimized. Traditionally, RA problems are nonconvex optimizations, which are performed using heuristic methods, such as genetic algorithm, particle swarm optimization, and simulated annealing. However, the application of these approaches remains computationally expensive and unattractive for dense cellular networks. Therefore, artificial intelligence algorithms are used to improve traditional RA mechanisms. Deep learning is a promising tool for addressing resource management problems in wireless communication. In this study, we investigate a double deep Q-network-based RA framework that maximizes energy efficiency (EE) and total network throughput in unmanned aerial vehicle (UAV)-assisted terrestrial networks. Specifically, the system is studied under the constraints of interference. However, the optimization problem is formulated as a mixed integer nonlinear program. Within this framework, we evaluated the effect of height and the number of UAVs on EE and throughput. Then, in accordance with the experimental results, we compare the proposed algorithm with several artificial intelligence methods. Simulation results indicate that the proposed approach can increase EE with a considerable throughput.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>UAV</kwd>
<kwd>terrestrial network</kwd>
<kwd>reinforcement learning</kwd>
<kwd>mmWave</kwd>
<kwd>resource allocation</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label><title>Introduction</title>
<p>In recent years, unmanned aerial vehicle (UAV)-assisted fifth-generation (5G) communication has provided an attractive way to connect users with different devices and improve network capacity. However, data traffic on cellular networks increases exponentially, Thus, resource allocation (RA) is becoming increasingly critical [<xref ref-type="bibr" rid="ref-1">1</xref>]. Industrial spectrum bands experience increased demand for channels, leading to a spectrum scarcity situation. In the context of 5G, mmWave is considered a potential solution to meet this demand [<xref ref-type="bibr" rid="ref-2">2</xref>,<xref ref-type="bibr" rid="ref-3">3</xref>]. Moreover, other techniques, such as beamforming, multi-input multi-output (MIMO), and advanced power control, are introduced as promising solutions in the design of future networks [<xref ref-type="bibr" rid="ref-4">4</xref>]. Despite all these attempts to satisfy this demand, RA remains a priority to accommodate users in terms of Quality of Service (QoS). RA problems are often formulated as nonconvex problems requiring proper management [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-6">6</xref>]. Optimal solutions are obtained by implementing heuristic methods, such as genetic algorithm, particle swarm optimization, and simulated annealing [<xref ref-type="bibr" rid="ref-7">7</xref>,<xref ref-type="bibr" rid="ref-8">8</xref>]. However, such solutions end up with quasioptimal solutions and converge relatively slowly. Therefore, alternative solutions and flexible algorithms that exploit late development in artificial intelligence are desirable to explore. Recently, deep learning (DL) [<xref ref-type="bibr" rid="ref-9">9</xref>] has emerged as an effective tool to increase flexibility and optimize RA in complex wireless communication networks. First, DL-based RA is flexible because the same deep neural network (DNN) can be implemented to achieve different design objectives by modifying the loss function [<xref ref-type="bibr" rid="ref-10">10</xref>]. Second, the computation time required by DL to obtain RA results is lower than that of conventional algorithms [<xref ref-type="bibr" rid="ref-11">11</xref>]. Finally, DL can receive complex high-dimensional information as input and allocate the optimal action for each input statistic in a particular condition [<xref ref-type="bibr" rid="ref-10">10</xref>]. On the basis of the above analysis, DL can be chosen as an accurate method for RA.</p>
<sec id="s1_1">
<label>1.1</label><title>Related Works</title>
<p>As an emerging technology, DL has been used in several research studies to improve RA for terrestrial networks. For instance, the authors in [<xref ref-type="bibr" rid="ref-12">12</xref>] investigated the deep reinforcement learning (DRL)-based time division duplex configuration to allocate radio resources dynamically in an online manner and with high mobility. In [<xref ref-type="bibr" rid="ref-1">1</xref>], Lee et al. proposed deep power control based on a convolutional neural network to maximize spectral efficiency (SE) and energy efficiency (EE). In this study, a comparison between the DL model and a conventional weighted minimum mean square error was realized. In the same context, [<xref ref-type="bibr" rid="ref-13">13</xref>] performed a max-min and max-prod power allocation in downlink massive MIMO. To maximize EE, a deep artificial neural network scheme was applied in [<xref ref-type="bibr" rid="ref-14">14</xref>], where interference and system propagation channels were considered. Deep Q-learning (DQL)-based RA has also attracted much attention in recent literature. In [<xref ref-type="bibr" rid="ref-15">15</xref>], the authors studied the RA problem to enhance EE. The proposed method formulated a combined optimization problem, considering EE and QoS. More recently, a supervised DL approach in 5G multitier networks was adopted in [<xref ref-type="bibr" rid="ref-16">16</xref>] to solve the joint RA and remote&#x2013;radio&#x2013;head association. For this model, efficient subchannel and power allocation were used to generate training data. According to the decentralized RA mechanism, the authors in [<xref ref-type="bibr" rid="ref-17">17</xref>] developed a novel decentralized DL for vehicle-to-vehicle communications. The main objective was to determine the optimal sub-band and power level for transmission without requiring or waiting for global information. The authors used a DRL-based power control to investigate the problem of spectrum sharing in a cognitive radio system. The aim of this framework is that the secondary user shares the common spectrum with the primary user. Instead of unsupervised learning, the authors in [<xref ref-type="bibr" rid="ref-18">18</xref>] introduced supervised learning to maximize the throughput of device-to-device with maximum power constraint. The authors in [<xref ref-type="bibr" rid="ref-19">19</xref>] presented a comprehensive approach and considered DRL to maximize the total network throughput. However, this work did not include EE for optimization. Majority of the learning algorithms introduced above do not incorporate constraints directly into the training cost functions. Nowadays, literatures focus on RA in UAV-assisted cellular networks based on artificial intelligence. In reference [<xref ref-type="bibr" rid="ref-20">20</xref>], authors proposed a multiagent reinforcement learning framework to study the dynamic RA of multiple UAVs. The objective of this investigation was to maximize long-term rewards. However, the work did not consider UAV height. In [<xref ref-type="bibr" rid="ref-21">21</xref>], the authors used deep Q-network to solve the RA for UAV-assisted ultradense networks. To maximize system EE, a link selection strategy was proposed to allow users to select the optimal communication links. The authors did not consider the influence of SE on EE. In addition, the authors in [<xref ref-type="bibr" rid="ref-22">22</xref>] studied the RA problem of UAV-assisted wireless-powered Internet of Things systems, aiming to allocate optimal energy resources for wireless power transfer. In [<xref ref-type="bibr" rid="ref-23">23</xref>], the authors thoroughly investigated deep Q-network (DQN), invoking the difference of convex-based optimization method for multicooperative UAV-assisted wireless networks. This work assumed beamforming technique to serve users simultaneously in the same spectrum and maximize the sum user achievable rate. However, the work was not focused on EE. Another DQN work in [<xref ref-type="bibr" rid="ref-24">24</xref>] was presented to study the low utilization rate of resources. A novel DQN-based method was introduced to address the complex problem. The authors in [<xref ref-type="bibr" rid="ref-25">25</xref>] analyzed RA for bandwidth, throughput, and power consumption in different scenarios for multi-UAV-assisted IoT networks. On the basis of machine learning, authors considered DRL to address the joint RA problem. Although the proposed approach remained efficient, it did not consider ground network, EE and total throughput. In the present study, we aim to optimize EE and total throughput in UAV-assisted terrestrial networks subject to the constraints on transmission power and UAV height. Our main efforts are to apply a double deep Q-network (DDQN) that obtains optimal rewards better than DQN [<xref ref-type="bibr" rid="ref-26">26</xref>].</p>
</sec>
<sec id="s1_2">
<label>1.2</label><title>Contribution</title>
<p>Existing research on RA in UAV-assisted 5G networks focuses on single objective optimization and considers the DQN algorithm to generate data. Following the previous analysis, we investigate the RA problem in UAV-assisted cellular networks that maximize EE and total network throughput. Especially, DDQN is proposed to address intelligent RA. The main contributions of this study are listed below.<list list-type="simple"><list-item><label>(1)</label>
<p>We formulate EE and total throughput in mmWave scenario while ensuring the minimum QoS requirements for all users according to the environment. However, the optimization problem is formulated as a mixed integer nonlinear program. Multiple constraints, such as the path loss model, number of users, channel gains, beamforming, and signal-to-interference-plus-noise ratio (SINR) issues, are used to describe the environment.</p></list-item><list-item><label>(2)</label>
<p>We investigate a multiagent DDQN algorithm to optimize EE and total throughput. We assume that each user equipment (UE) behaves as an agent and performs optimization decisions on environmental information.</p></list-item><list-item><label>(3)</label>
<p>We compare the performance of the proposed algorithm, QL, and the DQN approaches already proposed in terms of RA.</p></list-item></list></p>
<p>The remainder of this paper is organized as follows: An overview for DRL is presented in Section 2, and the system model is introduced in Section 3. Then, the DDQN algorithm is discussed in Section 4, followed by simulation and results in Section 5. Lastly, conclusions and perspectives are drawn in Section 6.</p>
</sec>
</sec>
<sec id="s2">
<label>2</label><title>Overview of DRL</title>
<p>DRL is a prominent case of machine learning and thus a class of artificial intelligence. It allows agents to identify the ideal performance based on its own experience, rather than depending on a supervisor [<xref ref-type="bibr" rid="ref-27">27</xref>]. In this approach, a neural network is used as an agent that learns by interacting with the environment and solves the process by determining an optimal action. Compared with the standard ML, namely supervised and unsupervised learning [<xref ref-type="bibr" rid="ref-28">28</xref>], DRL does not depend on data acquisition. Thus, sequential decision making occurs, and the next input is based on the decision of the learner or system. Moreover, in DRL, the Markov decision process (MDP) is formalized as a mathematical approach to modeling and decision-making situations. The reinforcement learning process operates as follows [<xref ref-type="bibr" rid="ref-29">29</xref>]: the agent begins in a specific state within its environment <inline-formula id="ieqn-1">
<mml:math id="mml-ieqn-1"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mi>S</mml:mi></mml:math>
</inline-formula> by obtaining an initial observation <inline-formula id="ieqn-2">
<mml:math id="mml-ieqn-2"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mi mathvariant="normal">&#x03A9;</mml:mi></mml:math>
</inline-formula> and takes an action <inline-formula id="ieqn-3">
<mml:math id="mml-ieqn-3"><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mi>A</mml:mi></mml:math>
</inline-formula> at each time step <inline-formula id="ieqn-4">
<mml:math id="mml-ieqn-4"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:math>
</inline-formula>. As illustrated in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>, the DRL can be categorized into three algorithms, such as value-based, policy gradient, and model-based methods. In value-based DRL, the agent uses the learned value function to evaluate <inline-formula id="ieqn-5">
<mml:math id="mml-ieqn-5"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>a</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</inline-formula> pairs and generate a policy [<xref ref-type="bibr" rid="ref-30">30</xref>]. DQL is a much more popular and efficient algorithm in this category. By contrast, a policy-based algorithm is intuitive, where algorithms learn a policy <inline-formula id="ieqn-6">
<mml:math id="mml-ieqn-6"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>&#x03C0;</mml:mi></mml:math>
</inline-formula>. Learning a policy to act in an environment is sensible; thus, a policy function <inline-formula id="ieqn-7">
<mml:math id="mml-ieqn-7"><mml:mi>&#x03C0;</mml:mi></mml:math>
</inline-formula> considers a state <italic>s</italic> as input to generate an action <inline-formula id="ieqn-8">
<mml:math id="mml-ieqn-8"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>a</mml:mi><mml:mo>&#x223C;</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>s</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math>
</inline-formula>.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption><title>DRL algorithms</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-1.tif"/>
</fig>
<sec id="s2_1">
<label>2.1</label><title>Q-Learning</title>
<p>As a popular branch of machine learning, Q-learning is based on the main concept of the action value function <inline-formula id="ieqn-9">
<mml:math id="mml-ieqn-9"><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x03C0;</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>a</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</inline-formula> for policy <inline-formula id="ieqn-10">
<mml:math id="mml-ieqn-10"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>&#x03C0;</mml:mi></mml:math>
</inline-formula>. It uses the Bellman equation to learn and calculate the optimum values of the Q-function in an iterative way [<xref ref-type="bibr" rid="ref-30">30</xref>], which is expressed as<disp-formula id="eqn-1"><label>(1)</label>
<mml:math id="mml-eqn-1" display="block"><mml:mi>Q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo stretchy="false">&#x2190;</mml:mo><mml:mi>Q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msubsup><mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mi>Q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>Q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
<p>where <inline-formula id="ieqn-11">
<mml:math id="mml-ieqn-11"><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math>
</inline-formula> is the step-size parameter that defines the extent to which the new data contribute to the existing Q value, <inline-formula id="ieqn-12">
<mml:math id="mml-ieqn-12"><mml:mi>&#x03B3;</mml:mi></mml:math>
</inline-formula> is the MDP discounter factor, <inline-formula id="ieqn-13">
<mml:math id="mml-ieqn-13"><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math>
</inline-formula> is the numerical reward for the agent after the execution of the action, and <inline-formula id="ieqn-14">
<mml:math id="mml-ieqn-14"><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math>
</inline-formula> indicates that the environment changes to a new state, with transition probability <inline-formula id="ieqn-15">
<mml:math id="mml-ieqn-15"><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>r</mml:mi><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>a</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:math>
</inline-formula>, as illustrated in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>. However, the Q-learning algorithm could be applied only to RA problems with low dimensionality in state and action, resulting in an evolutionary limitation [<xref ref-type="bibr" rid="ref-31">31</xref>]. Moreover, this application is only used when the state and action spaces are discrete (e.g., channel access) [<xref ref-type="bibr" rid="ref-32">32</xref>].</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption><title>Q-learning algorithm for UAV-assisted SBS</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-2.tif"/>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label><title>Deep Q-Network</title>
<p>As stated above, the Q-learning algorithm faces difficulties in obtaining the optimal policy when the action and state spaces become exceptionally large [<xref ref-type="bibr" rid="ref-33">33</xref>]. This constraint is often observed in the RA approaches of cellular networks. To solve this problem, the DQN algorithm, which connects the traditional Q-learning algorithm to a convolutional neural network, was proposed [<xref ref-type="bibr" rid="ref-34">34</xref>]. The main difference with the Q-learning algorithm is the replacement of the table with the function approximator called DNN; this process attempts to approximate the Q values. Approximators have two types: a linear function and a nonlinear function [<xref ref-type="bibr" rid="ref-35">35</xref>]. However, in a nonlinear DNN, the new Q function is defined as <inline-formula id="ieqn-16">
<mml:math id="mml-ieqn-16"><mml:mtext>&#x00A0;</mml:mtext><mml:mi>Q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mi>&#x03C9;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2248;</mml:mo><mml:msup><mml:mi>Q</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>a</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</inline-formula>, where <inline-formula id="ieqn-17">
<mml:math id="mml-ieqn-17"><mml:mi>&#x03C9;</mml:mi></mml:math>
</inline-formula> represents the weights of the neural network. At each time <inline-formula id="ieqn-18">
<mml:math id="mml-ieqn-18"><mml:mtext>&#x00A0;</mml:mtext><mml:mi>t</mml:mi></mml:math>
</inline-formula>, action <inline-formula id="ieqn-19">
<mml:math id="mml-ieqn-19"><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math>
</inline-formula> is taken in accordance with the <inline-formula id="ieqn-20">
<mml:math id="mml-ieqn-20"><mml:mi>&#x03F5;</mml:mi></mml:math>
</inline-formula>-greedy policy, and the transition tuple (<inline-formula id="ieqn-21">
<mml:math id="mml-ieqn-21"><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math>
</inline-formula> is stored mainly in a replay memory denoted by <italic>D</italic>. During the training process, a minibatch is sampled randomly from experience <italic>D</italic> to optimize the mean squared error. Thus, the target Q-network is used to improve the stability of DQN, whose <inline-formula id="ieqn-22">
<mml:math id="mml-ieqn-22"><mml:mi>&#x03C9;</mml:mi></mml:math>
</inline-formula> is regularly adjusted to follow those of the principal Q-network. On the basis of the Bellman equation, the optimal state-action function is given by [<xref ref-type="bibr" rid="ref-35">35</xref>]<disp-formula id="eqn-2"><label>(2)</label>
<mml:math id="mml-eqn-2" display="block"><mml:msup><mml:mi>Q</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:msubsup><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:mi>r</mml:mi><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:msubsup><mml:mrow></mml:mrow><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:msup><mml:mi>Q</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">]</mml:mo><mml:mo>.</mml:mo></mml:math>
</disp-formula></p>
<p>To train the DQN, iterative updating of the weight <inline-formula id="ieqn-23">
<mml:math id="mml-ieqn-23"><mml:mi>&#x03C9;</mml:mi></mml:math>
</inline-formula> is used, thus minimizing the mean squared error of the Bellman equation. Mathematically, the loss function at each iteration is given by<disp-formula id="eqn-3"><label>(3)</label>
<mml:math id="mml-eqn-3" display="block"><mml:mi>L</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>&#x03C9;</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mi>D</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo><mml:mrow><mml:mi>&#x03B3;</mml:mi><mml:msubsup><mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mi>Q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mi>&#x03C9;</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo><mml:mrow><mml:mi>Q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mi>&#x03C9;</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:math>
</disp-formula></p>
</sec>
</sec>
<sec id="s3">
<label>3</label><title>System Model and Problem Formulation</title>
<p>In the proposed model, we consider the downlink communication of a UAV-assisted cellular network comprising a set of small base stations (SBSs) denoted as <inline-formula id="ieqn-24">
<mml:math id="mml-ieqn-24"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mi>M</mml:mi></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
</inline-formula> and a set of UAVs which is defined as <inline-formula id="ieqn-25">
<mml:math id="mml-ieqn-25"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>U</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:msub><mml:mi>V</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:msub><mml:mi>V</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:msub><mml:mi>V</mml:mi><mml:mi>u</mml:mi></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
</inline-formula>. UAVs are placed at a particular altitude <inline-formula id="ieqn-26">
<mml:math id="mml-ieqn-26"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>H</mml:mi></mml:math>
</inline-formula>, and <inline-formula id="ieqn-27">
<mml:math id="mml-ieqn-27"><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2264;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> is assumed to be constant for all UAVs. Each cell contains an mm-wave band and some user <italic>N</italic> distributed randomly in a dense area. We assume that a particular user is assigned to a single base station that provides the strongest signal. In this work, MBS and UEs are assumed to be equipped with omnidirectional antennas, i.e., antennas with unit gain, and every UAV are equipped with directional antennas [<xref ref-type="bibr" rid="ref-36">36</xref>]. Moreover, each UE associated with UAVs are assigned with orthogonal resource blocks RBs (an RB consists of 12 subcarriers, with a total bandwidth of 180&#x2005;kHz in the frequency domain and one time slot 0.5&#x2005;ms in the time domain), whereas UEs associated with SBSs share the remaining RBs [<xref ref-type="bibr" rid="ref-37">37</xref>]. The transmission power allocated by UAVs and SBSs are denoted by <inline-formula id="ieqn-28">
<mml:math id="mml-ieqn-28"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula>, <inline-formula id="ieqn-29">
<mml:math id="mml-ieqn-29"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> respectively. Furthermore, the link between BS &#x003D; UAVs <inline-formula id="ieqn-30">
<mml:math id="mml-ieqn-30"><mml:mrow><mml:mo>&#x222A;</mml:mo></mml:mrow></mml:math>
</inline-formula> SBSs and users can have two conditions, i.e., line-of-sight (LoS) or non-line-of-sight (NLoS) link. As illustrated in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, interference powers from adjacent base stations are considered. <xref ref-type="table" rid="table-1">Table 1</xref> summarizes the notations that were used in this article.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption><title>DQN architecture for UAV-assisted SBS</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-3.tif"/>
</fig><table-wrap id="table-1"><label>Table 1</label>
<caption><title>Summary of symbols and notations</title></caption>
<table><colgroup><col align="left"/><col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Symbol</th>
<th align="left">Definition</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left"><inline-formula id="ieqn-166">
<mml:math id="mml-ieqn-166"><mml:mi>&#x03C9;</mml:mi></mml:math>
</inline-formula></td>
<td align="left">Weights of the neural network</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-167">
<mml:math id="mml-ieqn-167"><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math>
</inline-formula></td>
<td align="left">State, action, and reward, respectively</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-168">
<mml:math id="mml-ieqn-168"><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow></mml:math>
</inline-formula>, <inline-formula id="ieqn-169">
<mml:math id="mml-ieqn-169"><mml:mi>U</mml:mi></mml:math>
</inline-formula></td>
<td align="left">Set of SBS and UAV</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-170">
<mml:math id="mml-ieqn-170"><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula>, <inline-formula id="ieqn-171">
<mml:math id="mml-ieqn-171"><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula></td>
<td align="left">Minimum and maximum altitude of UAV</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-172">
<mml:math id="mml-ieqn-172"><mml:mi>N</mml:mi></mml:math>
</inline-formula></td>
<td align="left">Number of users</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-173">
<mml:math id="mml-ieqn-173"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula></td>
<td align="left">Power transmission of UAV and SBS</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-174">
<mml:math id="mml-ieqn-174"><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula></td>
<td align="left">Channel gain from BS to user <italic>k</italic> on different subcarriers <inline-formula id="ieqn-175">
<mml:math id="mml-ieqn-175"><mml:mi>n</mml:mi></mml:math>
</inline-formula></td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-176">
<mml:math id="mml-ieqn-176"><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula></td>
<td align="left">Throughput of user <italic>k</italic> on the <inline-formula id="ieqn-177">
<mml:math id="mml-ieqn-177"><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> subcarrier</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-178">
<mml:math id="mml-ieqn-178"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula></td>
<td align="left">Blockage probability in LoS condition</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-179">
<mml:math id="mml-ieqn-179"><mml:mi>&#x03B2;</mml:mi></mml:math>
</inline-formula></td>
<td align="left">Blockage parameter</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-180">
<mml:math id="mml-ieqn-180"><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula>, <inline-formula id="ieqn-181">
<mml:math id="mml-ieqn-181"><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula></td>
<td align="left">SINR of UAV and SBS, respectively</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-182">
<mml:math id="mml-ieqn-182"><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula>, <inline-formula id="ieqn-183">
<mml:math id="mml-ieqn-183"><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula></td>
<td align="left">Directional beamforming gain</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-184">
<mml:math id="mml-ieqn-184"><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:math>
</inline-formula></td>
<td align="left">Additive white gaussian noise</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-185">
<mml:math id="mml-ieqn-185"><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula></td>
<td align="left">Path-loss exponent for LoS and NLoS</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-186">
<mml:math id="mml-ieqn-186"><mml:msubsup><mml:mi>&#x03B4;</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>&#x03B4;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula></td>
<td align="left">Additional losses</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-187">
<mml:math id="mml-ieqn-187"><mml:mi>E</mml:mi><mml:mi>E</mml:mi></mml:math>
</inline-formula>, <inline-formula id="ieqn-188">
<mml:math id="mml-ieqn-188"><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:math>
</inline-formula></td>
<td align="left">Energy efficiency and spectral efficiency, respectively</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-189">
<mml:math id="mml-ieqn-189"><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msubsup></mml:math>
</inline-formula>, <inline-formula id="ieqn-190">
<mml:math id="mml-ieqn-190"><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msubsup></mml:math>
</inline-formula></td>
<td align="left">SINR threshold for UAV and SBS</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-191">
<mml:math id="mml-ieqn-191"><mml:msubsup><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>Q</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msubsup></mml:math>
</inline-formula></td>
<td align="left">Data rate requirement</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-192">
<mml:math id="mml-ieqn-192"><mml:msubsup><mml:mi>r</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup></mml:math>
</inline-formula></td>
<td align="left">Reward function</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3_1">
<label>3.1</label><title>Fading and Achievable Data Rate</title>
<p>The channel between the base station and UE can be fixed or time varying. Fading is defined as the fluctuation in received signal strength with respect to time, and it occurs due to several factors, including transmitter and receiver movement, propagation environment, and atmospheric condition. Similar to [<xref ref-type="bibr" rid="ref-38">38</xref>], we model the channel in a way that it can capture small-scale and large-scale fading. At each time slot <inline-formula id="ieqn-31">
<mml:math id="mml-ieqn-31"><mml:mtext>&#x00A0;</mml:mtext><mml:mi>t</mml:mi></mml:math>
</inline-formula>, the small-scale fading between UAVs, SBSs, and UEs is considered frequency-selective fading, whose objective is to obtain a delay spread greater than the symbol period. By contrast, the channel in every subcarrier is supposed to be flat fading. This combination means that the channel gains can remain unchanged. All UEs periodically transmit their channel quality information to the related BS. In addition, let <inline-formula id="ieqn-32">
<mml:math id="mml-ieqn-32"><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> designate the channel gain from BS to user <italic>k</italic> on different subcarriers <inline-formula id="ieqn-33">
<mml:math id="mml-ieqn-33"><mml:mtext>&#x00A0;</mml:mtext><mml:mi>n</mml:mi></mml:math>
</inline-formula>. A binary variable <inline-formula id="ieqn-34">
<mml:math id="mml-ieqn-34"><mml:mi>&#x03C6;</mml:mi></mml:math>
</inline-formula> is introduced to define the association mode. If UE is associated with the UAV/SBS according to LoS link, then <inline-formula id="ieqn-35">
<mml:math id="mml-ieqn-35"><mml:mtext>&#x00A0;</mml:mtext><mml:mi>&#x03C6;</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math>
</inline-formula>; otherwise <inline-formula id="ieqn-36">
<mml:math id="mml-ieqn-36"><mml:mtext>&#x00A0;</mml:mtext><mml:mi>&#x03C6;</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math>
</inline-formula>. We apply the following assumption in formulation. The mm-wave signal is affected by various factors, such as buildings in urban areas, making the link susceptible to effect blockage. Thus, the downlink achievable throughput (data rate) of user <italic>k</italic> on the <inline-formula id="ieqn-37">
<mml:math id="mml-ieqn-37"><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> subcarrier can be given by the following equation as<disp-formula id="eqn-4"><label>(4)</label>
<mml:math id="mml-eqn-4" display="block"><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math>
</disp-formula><disp-formula id="eqn-5"><label>(5)</label>
<mml:math id="mml-eqn-5" display="block"><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03C6;</mml:mi><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>S</mml:mi><mml:mi>I</mml:mi><mml:mi>N</mml:mi><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03C6;</mml:mi><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>S</mml:mi><mml:mi>I</mml:mi><mml:mi>N</mml:mi><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</disp-formula></p>
<p>where <inline-formula id="ieqn-38">
<mml:math id="mml-ieqn-38"><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula>, <inline-formula id="ieqn-39">
<mml:math id="mml-ieqn-39"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> are the blockage probabilities when the link between the UAV/SBS and UE is LoS; they are expressed as [<xref ref-type="bibr" rid="ref-39">39</xref>,<xref ref-type="bibr" rid="ref-40">40</xref>]<disp-formula id="eqn-6"><label>(6)</label>
<mml:math id="mml-eqn-6" display="block"><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>b</mml:mi><mml:mo>+</mml:mo><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>c</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>180</mml:mn></mml:mrow><mml:mi>&#x03C0;</mml:mi></mml:mfrac></mml:mrow><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mi>H</mml:mi><mml:mi>z</mml:mi></mml:mfrac></mml:mrow></mml:mstyle></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>b</mml:mi></mml:mstyle></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
</disp-formula></p>
<p>where <italic>b</italic> and <italic>c</italic> are constants that depend on the network environment, and <inline-formula id="ieqn-40">
<mml:math id="mml-ieqn-40"><mml:mi>z</mml:mi><mml:mo>=</mml:mo><mml:msqrt><mml:msup><mml:mi>R</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo>&#x2212;</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:msqrt></mml:math>
</inline-formula> is the Euclidean distance between the typical UE and UAV (see <xref ref-type="fig" rid="fig-4">Fig. 4</xref>).<disp-formula id="eqn-7"><label>(7)</label>
<mml:math id="mml-eqn-7" display="block"><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi>&#x03B2;</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math>
</disp-formula>where <inline-formula id="ieqn-41">
<mml:math id="mml-ieqn-41"><mml:mi>&#x03B2;</mml:mi></mml:math>
</inline-formula> is the blockage parameter that defines the average size of obstacles. Here, <italic>d</italic> corresponds to the distance between SBS and UE.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption><title>UAV-assisted terrestrial network (SBS)</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-4.tif"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label><title>SINR and Path Loss Model</title>
<p>Adding additional gain to the system remains necessary due to the propagation losses that occur at mm-wave frequencies. One of the main solutions proposed by several research for future wireless networks is beamforming [<xref ref-type="bibr" rid="ref-41">41</xref>]. The fundamental principle of beamforming is to control the direction of a wavefront toward the UE. According to [<xref ref-type="bibr" rid="ref-42">42</xref>], UAV and SBS serve UE through beamforming technology. In this manner, the SINR of UE from UAV at time slot <italic>t</italic> can be written as<disp-formula id="eqn-8"><label>(8)</label>
<mml:math id="mml-eqn-8" display="block"><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mrow><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>h</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mi>P</mml:mi><mml:msubsup><mml:mi>L</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:mrow><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:msup><mml:mi>V</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mfrac></mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mstyle></mml:math>
</disp-formula></p>
<p>where <inline-formula id="ieqn-42">
<mml:math id="mml-ieqn-42"><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula> represents the directional beamforming gain for the desired link, and <inline-formula id="ieqn-43">
<mml:math id="mml-ieqn-43"><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula> refers to additive white Gaussian noise. <inline-formula id="ieqn-44">
<mml:math id="mml-ieqn-44"><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> and <inline-formula id="ieqn-45">
<mml:math id="mml-ieqn-45"><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:msup><mml:mi>V</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math>
</inline-formula> represent the interference from the adjacent SBS and UAV, respectively, and are expressed as<disp-formula id="eqn-9"><label>(9)</label>
<mml:math id="mml-eqn-9" display="block"><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>h</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</disp-formula><disp-formula id="eqn-10"><label>(10)</label>
<mml:math id="mml-eqn-10" display="block"><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:msup><mml:mi>V</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:msup><mml:mi>V</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>h</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:msup><mml:mi>V</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:msup><mml:mi>V</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</disp-formula></p>
<p>Without loss of generality, different properties are displayed in terms of propagation. For air-to-ground communication, the path loss of LoS and NLoS links at time slot <italic>t</italic> can be experienced depending on additional path losses in LoS and NLoS links <inline-formula id="ieqn-46">
<mml:math id="mml-ieqn-46"><mml:msubsup><mml:mi>&#x03C6;</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;&#x00A0;</mml:mtext><mml:msubsup><mml:mi>&#x03C6;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula> and path loss exponents <inline-formula id="ieqn-47">
<mml:math id="mml-ieqn-47"><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula>, <inline-formula id="ieqn-48">
<mml:math id="mml-ieqn-48"><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula> as<disp-formula id="eqn-11"><label>(11)</label>
<mml:math id="mml-eqn-11" display="block"><mml:mi>P</mml:mi><mml:msubsup><mml:mi>L</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>&#x03C6;</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:msup><mml:mtext>&#x00A0;</mml:mtext><mml:mi>f</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>&#x03C6;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:msup><mml:mtext>&#x00A0;</mml:mtext><mml:mi>f</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</disp-formula></p>
<p>Similarly, we define the SINR when UE is associated with SBS. In this case, we adopt the standard power-law path loss model with the mean <inline-formula id="ieqn-49">
<mml:math id="mml-ieqn-49"><mml:msubsup><mml:mi>&#x03B4;</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula>, <inline-formula id="ieqn-50">
<mml:math id="mml-ieqn-50"><mml:msubsup><mml:mi>&#x03B4;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula> for LoS and NLoS, respectively. Hence, the path loss model can be given as<disp-formula id="eqn-12"><label>(12)</label>
<mml:math id="mml-eqn-12" display="block"><mml:mi>P</mml:mi><mml:msubsup><mml:mi>L</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>&#x03B4;</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msup><mml:mi>d</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msubsup><mml:mi>a</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:mrow></mml:msup><mml:mtext>&#x00A0;</mml:mtext><mml:mi>f</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>&#x03B4;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msup><mml:mi>d</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msubsup><mml:mi>a</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:mrow></mml:msup><mml:mtext>&#x00A0;</mml:mtext><mml:mi>f</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:math>
</disp-formula></p>
<p>The SINR additional loss at the typical UE when it is connected to SBS is given by <xref ref-type="disp-formula" rid="eqn-13">(13)</xref>, where <inline-formula id="ieqn-51">
<mml:math id="mml-ieqn-51"><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>h</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula> and <inline-formula id="ieqn-52">
<mml:math id="mml-ieqn-52"><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:msup><mml:mi>S</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:msup><mml:mi>S</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>h</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:msup><mml:mi>S</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:msup><mml:mi>S</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula> are the interferences from UAV and SBS, respectively.<disp-formula id="eqn-13"><label>(13)</label>
<mml:math id="mml-eqn-13" display="block"><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mrow><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>h</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mi>P</mml:mi><mml:msubsup><mml:mi>L</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:mrow><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:msup><mml:mi>S</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mfrac></mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mstyle></mml:math>
</disp-formula></p>
</sec>
<sec id="s3_3">
<label>3.3</label><title>Spectral Efficiency and Energy Efficiency</title>
<p>SE and EE are the key metrics to evaluate any wireless communication system. SE is defined as the efficiency capability of a given channel bandwidth. In other words, it represents the transmission rate per unit of bandwidth and is measured in bits per second per hertz. The EE metric is used to evaluate the total energy consumption for a network. It is defined as a ratio of the total transferred bits to the total power consumption. Nevertheless, EE and SE have a fundamental relationship. Let <inline-formula id="ieqn-53">
<mml:math id="mml-ieqn-53"><mml:msub><mml:mi>P</mml:mi><mml:mi>C</mml:mi></mml:msub><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula> be the power consumed in the circuit of the transmitter; then, EE can be given by<disp-formula id="eqn-14"><label>(14)</label>
<mml:math id="mml-eqn-14" display="block"><mml:mi>E</mml:mi><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>S</mml:mi><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>C</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
</disp-formula>where <inline-formula id="ieqn-54">
<mml:math id="mml-ieqn-54"><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math>
</inline-formula> is the transmit power <inline-formula id="ieqn-55">
<mml:math id="mml-ieqn-55"><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
</inline-formula>, which ranges <inline-formula id="ieqn-56">
<mml:math id="mml-ieqn-56"><mml:mtext>&#x00A0;</mml:mtext><mml:mn>0</mml:mn><mml:mo>&#x003C;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula>; the achievable <inline-formula id="ieqn-57">
<mml:math id="mml-ieqn-57"><mml:mi>S</mml:mi><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:math>
</inline-formula> of transmitter can be computed as</p>
<p><bold><underline>For UAV</underline></bold><disp-formula id="eqn-15"><label>(15)</label>
<mml:math id="mml-eqn-15" display="block"><mml:mi>S</mml:mi><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mrow><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>h</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mi>P</mml:mi><mml:msubsup><mml:mi>L</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:mrow><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:msup><mml:mi>V</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
<p><bold><underline>For SBS</underline></bold><disp-formula id="eqn-16"><label>(16)</label>
<mml:math id="mml-eqn-16" display="block"><mml:mi>S</mml:mi><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mrow><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>h</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:msubsup><mml:mi>G</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mi>P</mml:mi><mml:msubsup><mml:mi>L</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup></mml:mrow><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:msup><mml:mi>S</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
</sec>
<sec id="s3_4">
<label>3.4</label><title>Objective Formulation</title>
<p>The proper performance of EE approaches is of paramount importance in UAV-assisted terrestrial networks because it is directly related to the choice of objectives and constraints for relevant optimization problems. In this work, we aim to optimize two specific objectives for RA, namely, the maximization of EE and throughput. From the SE perspective, the EE maximization problem can be formulated as<disp-formula id="eqn-17"><label>(17)</label>
<mml:math id="mml-eqn-17" display="block"><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:munder><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:munder><mml:mo>&#x2061;</mml:mo><mml:mi>E</mml:mi><mml:msup><mml:mi>E</mml:mi><mml:mi>t</mml:mi></mml:msup></mml:math>
</disp-formula></p>
<p>s.t. <inline-formula id="ieqn-58">
<mml:math id="mml-ieqn-58"><mml:mi>C</mml:mi><mml:mn>1</mml:mn></mml:math>
</inline-formula>: <inline-formula id="ieqn-59">
<mml:math id="mml-ieqn-59"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mn>0</mml:mn><mml:mo>&#x003C;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">x</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>;</mml:mo><mml:mtext>&#x00A0;&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>U</mml:mi><mml:mo>;</mml:mo><mml:mtext>&#x00A0;&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula></p>
<p>&#x2003;&#x2002;<inline-formula id="ieqn-60">
<mml:math id="mml-ieqn-60"><mml:mi>C</mml:mi><mml:mn>2</mml:mn><mml:mo>&#x003A;</mml:mo><mml:mspace width="thinmathspace" /><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2264;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>U</mml:mi><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula></p>
<p>&#x2003;&#x2002;<inline-formula id="ieqn-61">
<mml:math id="mml-ieqn-61"><mml:mi>C</mml:mi><mml:mn>3</mml:mn><mml:mo>&#x003A;</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>E</mml:mi><mml:msubsup><mml:mi>E</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>&#x2265;</mml:mo><mml:mi>E</mml:mi><mml:msubsup><mml:mi>E</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>U</mml:mi><mml:mo>;</mml:mo><mml:mtext>&#x00A0;&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula></p>
<p>&#x2003;&#x2002;<inline-formula id="ieqn-62">
<mml:math id="mml-ieqn-62"><mml:mi>C</mml:mi><mml:mn>4</mml:mn></mml:math>
</inline-formula>: <inline-formula id="ieqn-63">
<mml:math id="mml-ieqn-63"><mml:msub><mml:mrow><mml:mi mathvariant="normal">R</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">B</mml:mi><mml:mi mathvariant="normal">S</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="normal">k</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="normal">n</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x2265;</mml:mo><mml:msubsup><mml:mi>R</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">B</mml:mi><mml:mi mathvariant="normal">S</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="normal">k</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="normal">n</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>Q</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula></p>
<p>&#x2003;&#x2002;<inline-formula id="ieqn-64">
<mml:math id="mml-ieqn-64"><mml:mi>C</mml:mi><mml:mn>5</mml:mn></mml:math>
</inline-formula>: <inline-formula id="ieqn-65">
<mml:math id="mml-ieqn-65"><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>&#x2265;</mml:mo><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msubsup><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>U</mml:mi><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula></p>
<p>&#x2003;&#x2002;<inline-formula id="ieqn-66">
<mml:math id="mml-ieqn-66"><mml:mi>C</mml:mi><mml:mn>6</mml:mn></mml:math>
</inline-formula>: <inline-formula id="ieqn-67">
<mml:math id="mml-ieqn-67"><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>&#x2265;</mml:mo><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msubsup><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula></p>
<p>&#x2003;&#x2002;<inline-formula id="ieqn-68">
<mml:math id="mml-ieqn-68"><mml:mi>C</mml:mi><mml:mn>7</mml:mn></mml:math>
</inline-formula>: <inline-formula id="ieqn-69">
<mml:math id="mml-ieqn-69"><mml:mi>&#x03BE;</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mn>1</mml:mn></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>U</mml:mi><mml:mo>;</mml:mo><mml:mtext>&#x00A0;&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula></p>
<p>Constraint <inline-formula id="ieqn-70">
<mml:math id="mml-ieqn-70"><mml:mi>C</mml:mi><mml:mn>1</mml:mn></mml:math>
</inline-formula> means that the transmit power <inline-formula id="ieqn-71">
<mml:math id="mml-ieqn-71"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> and <inline-formula id="ieqn-72">
<mml:math id="mml-ieqn-72"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> must be in the interval <inline-formula id="ieqn-73">
<mml:math id="mml-ieqn-73"><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math>
</inline-formula>. It specifies the upper limit of the power transmission. Constraint <inline-formula id="ieqn-74">
<mml:math id="mml-ieqn-74"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>C</mml:mi><mml:mn>2</mml:mn></mml:math>
</inline-formula> indicates that the UAV should be positioned between a minimum and maximum height. At higher heights, the distance between the UAV and UE increases, resulting in considerable path loss. By contrast, when the UAV is located at a certain minimum height, the NLoS conditions are recorded and may affect EE; hence, this constraint must be studied. The constraint in <inline-formula id="ieqn-75">
<mml:math id="mml-ieqn-75"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>C</mml:mi><mml:mn>3</mml:mn><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> guarantees that the EE of UAV must be greater than that of SBS. In <inline-formula id="ieqn-76">
<mml:math id="mml-ieqn-76"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>C</mml:mi><mml:mn>4</mml:mn></mml:math>
</inline-formula>, <inline-formula id="ieqn-77">
<mml:math id="mml-ieqn-77"><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula> defines the maximum downlink achievable data rate, whereas <inline-formula id="ieqn-78">
<mml:math id="mml-ieqn-78"><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>Q</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula> accounts for the data rate requirement. Constraints <inline-formula id="ieqn-79">
<mml:math id="mml-ieqn-79"><mml:mtext>&#x00A0;</mml:mtext><mml:mi>C</mml:mi><mml:mn>5</mml:mn><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula> and <inline-formula id="ieqn-80">
<mml:math id="mml-ieqn-80"><mml:mtext>&#x00A0;</mml:mtext><mml:mi>C</mml:mi><mml:mn>6</mml:mn><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula> specify that the SINR of the UE should be higher than a certain threshold; the SINR threshold differs from each tier (UAV, SBS). Lastly, the last constraint ensures that the UE is connected with a single BS. In a subsequent section, we will present our second objective, which is to maximize the total network throughput. The overall throughput <inline-formula id="ieqn-81">
<mml:math id="mml-ieqn-81"><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula> is defined as the sum of the data rates that are provided perfectly to all UE. Mathematically, the maximization problem can be computed as<disp-formula id="eqn-18"><label>(18)</label>
<mml:math id="mml-eqn-18" display="block"><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:munder><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:munder><mml:mo>&#x2061;</mml:mo><mml:msubsup><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</disp-formula></p>
<p>s.t. <inline-formula id="ieqn-82">
<mml:math id="mml-ieqn-82"><mml:mi>C</mml:mi><mml:mn>1</mml:mn></mml:math>
</inline-formula>: <inline-formula id="ieqn-83">
<mml:math id="mml-ieqn-83"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mn>0</mml:mn><mml:mo>&#x003C;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>;</mml:mo><mml:mtext>&#x00A0;&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>U</mml:mi><mml:mo>;</mml:mo><mml:mtext>&#x00A0;&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow></mml:math>
</inline-formula></p>
<p>&#x2003;&#x2002;<inline-formula id="ieqn-84">
<mml:math id="mml-ieqn-84"><mml:mi>C</mml:mi><mml:mn>2</mml:mn></mml:math>
</inline-formula>: <inline-formula id="ieqn-85">
<mml:math id="mml-ieqn-85"><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2264;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>U</mml:mi></mml:math>
</inline-formula></p>
<p>&#x2003;&#x2002;<inline-formula id="ieqn-86">
<mml:math id="mml-ieqn-86"><mml:mi>C</mml:mi><mml:mn>3</mml:mn></mml:math>
</inline-formula>: <inline-formula id="ieqn-87">
<mml:math id="mml-ieqn-87"><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msubsup><mml:mo>&#x2265;</mml:mo><mml:msubsup><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msubsup><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>U</mml:mi><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula></p>
<p>&#x2003;&#x2002;<inline-formula id="ieqn-88">
<mml:math id="mml-ieqn-88"><mml:mi>C</mml:mi><mml:mn>4</mml:mn><mml:mo>&#x003A;</mml:mo><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2265;</mml:mo><mml:msubsup><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>U</mml:mi><mml:mo>;</mml:mo><mml:mtext>&#x00A0;&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula></p>
<p>&#x2003;&#x2002;<inline-formula id="ieqn-89">
<mml:math id="mml-ieqn-89"><mml:mi>C</mml:mi><mml:mn>5</mml:mn></mml:math>
</inline-formula>: <inline-formula id="ieqn-90">
<mml:math id="mml-ieqn-90"><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>&#x003C;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>U</mml:mi><mml:mo>;</mml:mo><mml:mtext>&#x00A0;&#x00A0;</mml:mtext><mml:mi mathvariant="normal">&#x2200;</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula></p>
<p>The constraint in <inline-formula id="ieqn-91">
<mml:math id="mml-ieqn-91"><mml:mi>C</mml:mi><mml:mn>4</mml:mn></mml:math>
</inline-formula> indicates the minimum required data rate for QoS. Here, constraint <inline-formula id="ieqn-92">
<mml:math id="mml-ieqn-92"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>C</mml:mi><mml:mn>5</mml:mn><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> means that the LoS probability of the SBS must be less than that of the UAV.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label><title>Double Deep Q-Network Algorithm</title>
<p>In this section, we present a DRL algorithm-based EE and throughput RA framework to address the network problems of <xref ref-type="disp-formula" rid="eqn-17">(17)</xref> and <xref ref-type="disp-formula" rid="eqn-18">(18)</xref>. The task of the DRL agent is to learn an optimal policy from state to action, thus maximizing the utility function. We formulate the optimization problem as a fully observable Markov decision process. Similar to literature, we consider a tuple (<inline-formula id="ieqn-93">
<mml:math id="mml-ieqn-93"><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math>
</inline-formula>. Based on the transition probability <inline-formula id="ieqn-94">
<mml:math id="mml-ieqn-94"><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math>
</inline-formula>, the current network state <inline-formula id="ieqn-95">
<mml:math id="mml-ieqn-95"><mml:msub><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math>
</inline-formula> learns a new state according to the action <inline-formula id="ieqn-96">
<mml:math id="mml-ieqn-96"><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math>
</inline-formula> selected by the agent at time slot <inline-formula id="ieqn-97">
<mml:math id="mml-ieqn-97"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:math>
</inline-formula>. A DDQN is applied to achieve an optimal solution. However, we assume that UAV and SBS act as an agent that continuously interacts with the environment to optimize the policy. First, the agent <inline-formula id="ieqn-98">
<mml:math id="mml-ieqn-98"><mml:mi>j</mml:mi><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> observes the state <inline-formula id="ieqn-99">
<mml:math id="mml-ieqn-99"><mml:msubsup><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> and decides to take an action <inline-formula id="ieqn-100">
<mml:math id="mml-ieqn-100"><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> in accordance with the optimal policy. Then, at each time policy, the agent receives reward <inline-formula id="ieqn-101">
<mml:math id="mml-ieqn-101"><mml:msubsup><mml:mi>r</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> conditioned by the action and moves to the next state <inline-formula id="ieqn-102">
<mml:math id="mml-ieqn-102"><mml:msubsup><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>j</mml:mi></mml:msubsup></mml:math>
</inline-formula>. This procedure concerns the DQN algorithm with a single agent. The major inconvenience of this algorithm lies in the confusion of the selection or evaluation of actions, leading to overestimation of action values and unstable training. To solve this overestimation, Hasselt et al. proposed a DDQN architecture, where the max function estimators is decomposed into action selection and evaluation, as illustrated in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>. The fundamental concept of the algorithm is to change the target network <inline-formula id="ieqn-103">
<mml:math id="mml-ieqn-103"><mml:msubsup><mml:mi>Y</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>D</mml:mi><mml:mi>Q</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:msubsup><mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">a</mml:mi><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mi>Q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msubsup><mml:mi>&#x03C9;</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:msup><mml:mi></mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msubsup></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> as<disp-formula id="eqn-19"><label>(19)</label>
<mml:math id="mml-eqn-19" display="block"><mml:msubsup><mml:mi>Y</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>D</mml:mi><mml:mi>D</mml:mi><mml:mi>Q</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mi>Q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mrow></mml:mrow><mml:mrow><mml:msup><mml:mi>a</mml:mi><mml:mrow><mml:mo>&#x2217;</mml:mo><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mi>Q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msup><mml:mi>a</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msub><mml:mi>&#x03C9;</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>;</mml:mo><mml:mtext>&#x00A0;&#x00A0;</mml:mtext><mml:msubsup><mml:mi>&#x03C9;</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
<fig id="fig-5">
<label>Figure 5</label>
<caption><title>DDQN architecture</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-5.tif"/>
</fig>
<p>At each time <italic>t</italic>, the weighted parameters <inline-formula id="ieqn-104">
<mml:math id="mml-ieqn-104"><mml:msub><mml:mi>&#x03C9;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> of the online network is used to evaluate the greedy policy, whereas the weighted parameter <inline-formula id="ieqn-105">
<mml:math id="mml-ieqn-105"><mml:msubsup><mml:mi>&#x03C9;</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:msup><mml:mi></mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msubsup><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> estimates the policy value. For improved performance evaluation, the target network for DDQN can use any parameters from the previous iteration <inline-formula id="ieqn-106">
<mml:math id="mml-ieqn-106"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</inline-formula>. Therefore, a periodic update of the target network settings is applied with copies of the online network.</p>
<sec id="s4_1">
<label>4.1</label><title>State and Observation</title>
<p>The state describes a specific configuration of the environment. At time slot <inline-formula id="ieqn-107">
<mml:math id="mml-ieqn-107"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:math>
</inline-formula>, UAVs and SBSs act as agents and define the observation space <inline-formula id="ieqn-108">
<mml:math id="mml-ieqn-108"><mml:msubsup><mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow></mml:mrow><mml:mi>j</mml:mi><mml:mi>t</mml:mi></mml:msubsup></mml:math>
</inline-formula>. The observation of each <inline-formula id="ieqn-109">
<mml:math id="mml-ieqn-109"><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi><mml:mrow><mml:mo>&#x222A;</mml:mo></mml:mrow><mml:mo>&#x2061;</mml:mo><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:math>
</inline-formula> includes the SINR measurement from the UAV and SBS to UE, the height of UAVs <inline-formula id="ieqn-110">
<mml:math id="mml-ieqn-110"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>H</mml:mi></mml:math>
</inline-formula>, and spectral efficiency. We define the global state as<disp-formula id="eqn-20"><label>(20)</label>
<mml:math id="mml-eqn-20" display="block"><mml:msubsup><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow><mml:mi>t</mml:mi><mml:mn>1</mml:mn></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow><mml:mi>t</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x2026;</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow><mml:mi>t</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</disp-formula>where <inline-formula id="ieqn-111">
<mml:math id="mml-ieqn-111"><mml:msubsup><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">B</mml:mi><mml:mi mathvariant="normal">S</mml:mi></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> represents the set of observation and can be expressed as<disp-formula id="eqn-21"><label>(21)</label>
<mml:math id="mml-eqn-21" display="block"><mml:msubsup><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>I</mml:mi><mml:mi>N</mml:mi><mml:msubsup><mml:mi>R</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>H</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>S</mml:mi><mml:msubsup><mml:mi>E</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
</sec>
<sec id="s4_2">
<label>4.2</label><title>Action</title>
<p>In our problem, each agent must choose an appropriate base station (i.e., UAV or SBS), power transmission, UAV height, and LoS/NLoS link probability. At time step <inline-formula id="ieqn-112">
<mml:math id="mml-ieqn-112"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:math>
</inline-formula>, the action of UAV/SBS can be expressed as<disp-formula id="eqn-22"><label>(22)</label>
<mml:math id="mml-eqn-22" display="block"><mml:msubsup><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>b</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:msubsup><mml:mi>P</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>H</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mrow><mml:mi mathvariant="script">P</mml:mi></mml:mrow><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
<p>where <inline-formula id="ieqn-113">
<mml:math id="mml-ieqn-113"><mml:msubsup><mml:mi>b</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x2026;</mml:mo><mml:mi>&#x03B2;</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
</inline-formula> is the selected BS;<inline-formula id="ieqn-114">
<mml:math id="mml-ieqn-114"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msubsup><mml:mi>P</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x2026;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
</inline-formula> is the power transmission requirement, which indicates how much power should be assigned to UE; <inline-formula id="ieqn-115">
<mml:math id="mml-ieqn-115"><mml:msubsup><mml:mi>H</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x2026;</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
</inline-formula> is the UAV elevation.</p>
</sec>
<sec id="s4_3">
<label>4.3</label><title>Reward</title>
<p>Reinforcement learning is based on the reward function, stating that the agent (UAV and SBS) is guided toward an optimal policy. As mentioned above, we model this problem as a fully observable MDP to maximize EE and throughput. Therefore, the reward of <italic>j</italic> can be computed as<disp-formula id="eqn-23"><label>(23)</label>
<mml:math id="mml-eqn-23" display="block"><mml:msubsup><mml:mi>r</mml:mi><mml:mi>t</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>E</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>h</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi><mml:mi>h</mml:mi><mml:mi>p</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math>
</disp-formula>where <inline-formula id="ieqn-116">
<mml:math id="mml-ieqn-116"><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>E</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mo>+</mml:mo><mml:mi>u</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x2061;</mml:mo><mml:mi>E</mml:mi><mml:msup><mml:mi>E</mml:mi><mml:mi>t</mml:mi></mml:msup></mml:math>
</inline-formula> and <inline-formula id="ieqn-117">
<mml:math id="mml-ieqn-117"><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>h</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi><mml:mi>h</mml:mi><mml:mi>p</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mo>+</mml:mo><mml:mi>u</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x2061;</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula>. <inline-formula id="ieqn-118">
<mml:math id="mml-ieqn-118"><mml:msub><mml:mi>w</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> and <inline-formula id="ieqn-119">
<mml:math id="mml-ieqn-119"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> are the weights for each objective, respectively. The pseudo code for DDQN is outlined in Algorithm 1.</p>
<fig id="fig-14">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-14.tif"/>
</fig>
</sec>
</sec>
<sec id="s5">
<label>5</label><title>Simulation Results</title>
<p>This section discusses the simulation and results for EE and throughput in the downlink UAV-assisted terrestrial network comprising eight SBSs with a radius of 500&#x2005;m and five UAVs deployed randomly in the area. The cell contains <inline-formula id="ieqn-141">
<mml:math id="mml-ieqn-141"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mn>20</mml:mn><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> randomly distributed users and uses mm-wave bands. We assume that the maximum power transmission for SBS is <inline-formula id="ieqn-142">
<mml:math id="mml-ieqn-142"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>23</mml:mn><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>d</mml:mi><mml:mi>B</mml:mi><mml:mi>m</mml:mi></mml:math>
</inline-formula>, and different values of maximum <inline-formula id="ieqn-143">
<mml:math id="mml-ieqn-143"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> is shown in simulation. The path loss exponent in the LoS and NLoS links for the UAV and SBS have the values <inline-formula id="ieqn-144">
<mml:math id="mml-ieqn-144"><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>3</mml:mn></mml:math>
</inline-formula>, <inline-formula id="ieqn-145">
<mml:math id="mml-ieqn-145"><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mn>5</mml:mn></mml:math>
</inline-formula>, <inline-formula id="ieqn-146">
<mml:math id="mml-ieqn-146"><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:math>
</inline-formula>, and <inline-formula id="ieqn-147">
<mml:math id="mml-ieqn-147"><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>4</mml:mn></mml:math>
</inline-formula>. In addition, the power consumed in the circuit of the transmitter <inline-formula id="ieqn-148">
<mml:math id="mml-ieqn-148"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>40</mml:mn><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>d</mml:mi><mml:mi>B</mml:mi><mml:mi>m</mml:mi></mml:math>
</inline-formula>. The added white Gaussian noise <inline-formula id="ieqn-149">
<mml:math id="mml-ieqn-149"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mn>114</mml:mn><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>d</mml:mi><mml:mi>B</mml:mi><mml:mi>m</mml:mi></mml:math>
</inline-formula>. In the DDQN algorithm, the DNN of each agent is a four-layer fully connected neural network with two hidden layers of 64 and 32 neurons. Other simulation and DDQN parameters are listed in <xref ref-type="table" rid="table-2">Table 2</xref>. The simulation is realized using MATLAB (R2017a) running on a Dell PC (2.8 Ghz @ Intel Core i7-7600U, 16 GB). In our simulation, we consider <inline-formula id="ieqn-150">
<mml:math id="mml-ieqn-150"><mml:msub><mml:mi>w</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:mn>0.6</mml:mn></mml:math>
</inline-formula> and <inline-formula id="ieqn-151">
<mml:math id="mml-ieqn-151"><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>w</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:mn>0.4</mml:mn></mml:math>
</inline-formula>.</p>
<table-wrap id="table-2"><label>Table 2</label>
<caption><title>Simulation parameters</title></caption>
<table><colgroup><col align="left"/><col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Parameter</th>
<th align="left">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Mini batch</td>
<td align="left">64</td>
</tr>
<tr>
<td align="left">Learning rate</td>
<td align="left">0.01</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-193">
<mml:math id="mml-ieqn-193"><mml:mi>&#x03B3;</mml:mi></mml:math>
</inline-formula></td>
<td align="left">0.9</td>
</tr>
<tr>
<td align="left">Replay memory <italic>D</italic> capacity</td>
<td align="left">2000</td>
</tr>
<tr>
<td align="left"><inline-formula id="ieqn-194">
<mml:math id="mml-ieqn-194"><mml:mi>&#x03F5;</mml:mi></mml:math>
</inline-formula>-greedy</td>
<td align="left">0.05</td>
</tr>
<tr>
<td align="left">Antenna gain</td>
<td align="left">9 dBi</td>
</tr>
<tr>
<td align="left">Subchannel band</td>
<td align="left">75&#x2005;kHz</td>
</tr>
<tr>
<td align="left">SBS bandwidth</td>
<td align="left">30&#x2005;MHz</td>
</tr>
<tr>
<td align="left">UAV bandwidth</td>
<td align="left">45&#x2005;MHz</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s5_1">
<label>5.1</label><title>Energy Efficiency Analysis</title>
<p>In this subsection, we show some results of EE, which are obtained using DDNQ. For improved performance validation, we compare our proposed algorithm with the DQN and QL architectures. Moreover, the effect of UE demand, number of UAVs, and beamforming on maximum power <inline-formula id="ieqn-152">
<mml:math id="mml-ieqn-152"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> are discussed. In the simulation evaluation, the parameter values in <xref ref-type="table" rid="table-2">Table 2</xref> are used, unless otherwise specified. First, we evaluate the effect of UE demand on the EE for different algorithms in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>. A common observation in <xref ref-type="fig" rid="fig-6">Fig. 6</xref> is that increasing UE demand can lead to increased EE; however, from 60 Mbps, EE converges less quickly.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption><title>EE as a function of UE demand for different algorithms</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-6.tif"/>
</fig>
<p>This result is obtained because when UE demand increases considerably (<inline-formula id="ieqn-153">
<mml:math id="mml-ieqn-153"><mml:mn>60</mml:mn><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mi mathvariant="normal">M</mml:mi><mml:mi mathvariant="normal">b</mml:mi><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">s</mml:mi></mml:mrow><mml:mo>&#x003E;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math>
</inline-formula>, all algorithms (DDQN, DQN, and QL) aim to maximize network throughput, which requires high transmission power, causing reduced EE. Another comment from <xref ref-type="fig" rid="fig-6">Fig. 6</xref> is that the DDQN algorithm can outperform DQN and QL. This outcome is achieved because the agent selects a more appropriate Q value to estimate the action. This perfection is mainly due to the two separate estimators applied in DDQN. In other terms, the use of the opposite estimator is cost effective for obtaining unbiased Q-values. A proposed solution to the EE problem when UE demand increases is to add base stations. The increase in the number of UAVs has a remarkable effect on EE, as illustrated in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>. As the number of UAVs increases, EE improves because the number of users covered in UAV LoS increases.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption><title>EE as a function of number of UAVs for different algorithms</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-7.tif"/>
</fig>
<p>Moreover, <xref ref-type="fig" rid="fig-7">Fig. 7</xref> demonstrates that the DDQN algorithm outperforms DQN and QL by 13.3&#x0025; on EE because traditional RL algorithms use a one-actor network to train multiple agents; thus, conflicts between agents are recorded. Next, EE is plotted as a function of the number of UE for different UAV height (<inline-formula id="ieqn-154">
<mml:math id="mml-ieqn-154"><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> constraint), as shown in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>. Moreover, an increase in the number of UE results in EE degradation because of the increase in energy consumption. <xref ref-type="fig" rid="fig-8">Fig. 8</xref> also shows that UAV height can affect EE. Therefore, EE increases as <inline-formula id="ieqn-155">
<mml:math id="mml-ieqn-155"><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> increases because the increase in UAV height results in additional UEs in the LoS link condition, leading to an increase in the total number of bits transmitted.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption><title>EE as a function of number of users for different <inline-formula id="ieqn-165">
<mml:math id="mml-ieqn-165"><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext></mml:math>
</inline-formula> constraints</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-8.tif"/>
</fig>
<p>As the number of UE increases, the power assigned to UE declines. Therefore, the increase in height compensates this shortcoming. <xref ref-type="fig" rid="fig-9">Fig. 9</xref> shows EE <italic>vs.</italic> the maximum power of UAV <inline-formula id="ieqn-156">
<mml:math id="mml-ieqn-156"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> with and without beamforming. A common observation in <xref ref-type="fig" rid="fig-9">Fig. 9</xref> is that EE decreases by extending the maximum transmission power of the UAV due to the increased energy consumption by users. In addition, when the power of UAVs increases, the links between UAVs and UEs are in NLoS condition, thus reducing EE. This analysis is conducted with and without beamforming. As illustrated in <xref ref-type="fig" rid="fig-9">Fig. 9</xref>, applied beamforming improves EE in each algorithm (DDQN and DQN) because beamforming provides additional gains and can overcome mm-wave blockage constraints.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption><title>EE <italic>vs.</italic> maximum UAV power transmission with and without beamforming</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-9.tif"/>
</fig>
</sec>
<sec id="s5_2">
<label>5.2</label><title>Throughput Analysis</title>
<p>To validate the accuracy of our approach, we analyze the total throughput (second objective) according to the number of UAVs deployed, UAV height <inline-formula id="ieqn-157">
<mml:math id="mml-ieqn-157"><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula>, and beamforming. Considering the first scenario, <xref ref-type="fig" rid="fig-10">Fig. 10</xref> depicts the total throughput as a function of the number of UAVs. As the number of UAVs increases, the total throughput is enhanced. Thus, DDQN outperforms DQN and QL. However, this effect is mainly due to the increase in LoS links. The same figure also shows that the total throughput reaches a congestion level at a particular number of UAVs due to the rise in interference between UAVs.</p>
<fig id="fig-10">
<label>Figure 10</label>
<caption><title>Throughput as a function of number of UAVs with different algorithms</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-10.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig-11">Fig. 11</xref> illustrates the variation of throughput <italic>vs.</italic> UAV height in different AI algorithms. According to <xref ref-type="fig" rid="fig-11">Fig. 11</xref>, throughput increases with maximization of altitude <inline-formula id="ieqn-158">
<mml:math id="mml-ieqn-158"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> because at low altitude, the propagation condition is in NLoS, and interference between tiers is observed. By contrast, when the UAV height increases, the LoS condition occurs, resulting in reduced loss. Moreover, saturation is experienced from an altitude of 130&#x2005;m because as UAV height increases, the distance between the UAV and UE increases, leading to signal attenuation.</p>
<fig id="fig-11">
<label>Figure 11</label>
<caption><title>Throughput analysis as a function of UAV height</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-11.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig-12">Fig. 12</xref> shows the variation of the throughput <italic>vs.</italic> maximum UAV power. As expected, the total throughput increases as <inline-formula id="ieqn-159">
<mml:math id="mml-ieqn-159"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula> increases. <xref ref-type="fig" rid="fig-12">Fig. 12</xref> also reveals that DDQN achieves a maximum throughput of 582.7 Mbps with a maximum power of <inline-formula id="ieqn-160">
<mml:math id="mml-ieqn-160"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>35</mml:mn><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow></mml:math>
</inline-formula> dBm. By contrast, DQN achieves a maximum throughput of 269,234 Mbps at the same <inline-formula id="ieqn-161">
<mml:math id="mml-ieqn-161"><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mtext>&#x00A0;</mml:mtext></mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>V</mml:mi></mml:mrow></mml:msub></mml:math>
</inline-formula>. Again, the proposed DDQN algorithm outperforms DQN. Finally, we plot the throughput as a function of blockage parameter <inline-formula id="ieqn-162">
<mml:math id="mml-ieqn-162"><mml:mi>&#x03B2;</mml:mi></mml:math>
</inline-formula> for SBS when UAVs are assumed to be located at <inline-formula id="ieqn-163">
<mml:math id="mml-ieqn-163"><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>120</mml:mn></mml:math>
</inline-formula> m, as shown in <xref ref-type="fig" rid="fig-13">Fig. 13</xref>. When <inline-formula id="ieqn-164">
<mml:math id="mml-ieqn-164"><mml:mi>&#x03B2;</mml:mi></mml:math>
</inline-formula> increases, the total throughput of the network decreases. Therefore, with the increase in obstacle density, more UEs are served by NLoS conditions. In addition, <xref ref-type="fig" rid="fig-13">Fig. 13</xref> shows that the proposed DDQN scheme converges to highly satisfactory solutions compared with the other approaches because it handles interference perfectly.</p>
<fig id="fig-12">
<label>Figure 12</label>
<caption><title>Throughput as a function of UAV power with beamforming</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-12.tif"/>
</fig><fig id="fig-13">
<label>Figure 13</label>
<caption><title>Throughput <italic>vs.</italic> blockage parameter</title></caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CSSE_34461-fig-13.tif"/>
</fig>
</sec>
</sec>
<sec id="s6">
<label>6</label><title>Conclusion</title>
<p>In this study, we proposed a DDQN scheme for RA optimization in UAV-assisted terrestrial networks. The problem is formulated as EE and throughput maximization. Initially, we provided a general overview of deep reinforcement architectures. Then, we presented the network architecture where the base stations use the beamforming technique during transmission. The proposed EE and throughput were assessed under the number of UAVs, beamforming, maximum UAV power transmission, and blockage parameter. The algorithm accuracy of the obtained EE and throughput was demonstrated by a comparison with deep Q-network and Q-learning. Our results indicate that EE can be affected by the number of UAVs to be deployed in the coverage area, as well as the maximum altitude variation (constraint). Moreover, the use of beamforming can be cost effective in improving EE. Our investigation also revealed other useful conclusions. For throughput analysis, the blockage parameter has a dominant influence on the throughput, and an optimal value can be selected. In terms of convergences, our DDQN consistently outperforms DQN and QL. In future work, other issues can be explored and investigated. For instance, UAV mobility can be considered, and an optimal mobility model can be selected to maximize throughput. Interference coordination may also be introduced between tiers.</p>
</sec>
</body>
<back>
<ack>
<p>The authors would like to acknowledge the financial support received from Princess Nourah bint Abdulrahman University Researchers Supporting Project Number (PNURSP2022R323), Princess Nourah bint Abdulrahman University, Riyadh, Saudi Arabia and Taif University Researchers Supporting Project Number TURSP-2020/34), Taif, Saudi Arabia.</p>
</ack>
<sec><title>Funding Statement</title>
<p>This work was supported by <funding-source>Princess Nourah bint Abdulrahman University Researchers</funding-source> Supporting Project Number (<award-id>PNURSP2022R323</award-id>), <funding-source>Princess Nourah bint Abdulrahman University</funding-source>, Riyadh, Saudi Arabia, and Taif University Researchers Supporting Project Number <award-id>TURSP-2020/34</award-id>), Taif, Saudi Arabia.</p>
</sec>
<sec sec-type="COI-statement"><title>Conflicts of Interest</title>
<p>The authors declare that they have no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear"><title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>Z.</given-names> <surname>Sylia</surname></string-name>, <string-name><given-names>G.</given-names> <surname>C&#x00E9;dric</surname></string-name>, <string-name><given-names>O. M.</given-names> <surname>Amine</surname></string-name> and <string-name><given-names>K.</given-names> <surname>Abdelkrim</surname></string-name></person-group>, &#x201C;<article-title>Resource allocation in a multi-carrier cell using scheduler algorithms</article-title>,&#x201D; in <conf-name>4th Int. Conf. on Optimization and Applications (ICOA)</conf-name>, <conf-loc>Mohammedia, Morocco</conf-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>5</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T. O.</given-names> <surname>Olwal</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Djouani</surname></string-name> and <string-name><given-names>A. M.</given-names> <surname>Kurien</surname></string-name></person-group>, &#x201C;<article-title>A survey of resource management toward 5G radio access networks</article-title>,&#x201D; <source>IEEE Communications Surveys &#x0026; Tutorials</source>, vol. <volume>18</volume>, no. <issue>3</issue>, pp. <fpage>1656</fpage>&#x2013;<lpage>1686</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T. S.</given-names> <surname>Rappaport</surname></string-name></person-group>, &#x201C;<article-title>Millimeter wave mobile communications for 5G cellular: It will work!</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>1</volume>, pp. <fpage>335</fpage>&#x2013;<lpage>349</lpage>, <year>2013</year>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. A.</given-names> <surname>Ouamri</surname></string-name>, <string-name><given-names>M. E.</given-names> <surname>Ote&#x015F;teanu</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Isar</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Azni</surname></string-name></person-group>, &#x201C;<article-title>Coverage, handoff and cost optimization for 5G heterogeneous network</article-title>,&#x201D; <source>Physical Communication</source>, vol. <volume>39</volume>, pp. <fpage>1</fpage>&#x2013;<lpage>8</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Sun</surname></string-name>, <string-name><given-names>C.</given-names> <surname>She</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Yang</surname></string-name>, <string-name><given-names>T. Q. S.</given-names> <surname>Quek</surname></string-name>, Y. Li</person-group> <italic>et al.,</italic> &#x201C;<article-title>Optimizing resource allocation in the short blocklength regime for ultra-reliable and low-latency communications</article-title>,&#x201D; <source>IEEE Transactions on Wireless Communications</source>, vol. <volume>18</volume>, no. <issue>1</issue>, pp. <fpage>402</fpage>&#x2013;<lpage>415</lpage>, <year>Jan. 2019</year>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Hu</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Ozmen</surname></string-name>, <string-name><given-names>M. C.</given-names> <surname>Gursoy</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Schmeink</surname></string-name></person-group>, &#x201C;<article-title>Optimal power allocation for QoS-constrained downlink multi-user networks in the finite blocklength regime</article-title>,&#x201D; <source>IEEE Transactions on Wireless Communications</source>, vol. <volume>17</volume>, no. <issue>9</issue>, pp. <fpage>5827</fpage>&#x2013;<lpage>5840</lpage>, <year>Sept. 2018</year>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Zhu</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Xiao</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Cao</surname></string-name>, <string-name><given-names>D. O.</given-names> <surname>Wu</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>Joint Tx-Rx beamforming and power allocation for 5G millimeter-wave non-orthogonal multiple access networks</article-title>,&#x201D; <source>IEEE Transactions on Communications</source>, vol. <volume>67</volume>, no. <issue>7</issue>, pp. <fpage>5114</fpage>&#x2013;<lpage>5125</lpage>, July <year>2019</year>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. O.</given-names> <surname>Oladejo</surname></string-name> and <string-name><given-names>O. E.</given-names> <surname>Falowo</surname></string-name></person-group>, &#x201C;<article-title>Latency-aware dynamic resource allocation scheme for multi-tier 5G network: A network slicing-multitenancy scenario</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>8</volume>, pp. <fpage>74834</fpage>&#x2013;<lpage>74852</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Lei</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Yuan</surname></string-name>, <string-name><given-names>T. X.</given-names> <surname>Vu</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Chatzinotas</surname></string-name> and <string-name><given-names>B.</given-names> <surname>Ottersten</surname></string-name></person-group>, &#x201C;<article-title>Learning-based resource allocation: Efficient content delivery enabled by convolutional neural network</article-title>,&#x201D; in <conf-name>IEEE 20th Int. Workshop on Signal Processing Advances in Wireless Communications (SPAWC)</conf-name>, <conf-loc>Cannes, France</conf-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>5</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K. I.</given-names> <surname>Ahmed</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Tabassum</surname></string-name> and <string-name><given-names>E.</given-names> <surname>Hossain</surname></string-name></person-group>, &#x201C;<article-title>Deep learning for radio resource allocation in multi-cell networks</article-title>,&#x201D; <source>IEEE Network</source>, vol. <volume>33</volume>, no. <issue>6</issue>, pp. <fpage>188</fpage>&#x2013;<lpage>195</lpage>, Dec. <year>2019</year>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Liang</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Ye</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Yu</surname></string-name> and <string-name><given-names>G. Y.</given-names> <surname>Li</surname></string-name></person-group>, &#x201C;<article-title>Deep-learning-based wireless resource allocation with application to vehicular networks</article-title>,&#x201D; <source>Proceedings of the IEEE</source>, vol. <volume>108</volume>, no. <issue>2</issue>, pp. <fpage>341</fpage>&#x2013;<lpage>356</lpage>, <year>Feb. 2020</year>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Tang</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Zhou</surname></string-name> and <string-name><given-names>N.</given-names> <surname>Kato</surname></string-name></person-group>, &#x201C;<article-title>Deep reinforcement learning for dynamic uplink/downlink resource allocation in high mobility 5G HetNet</article-title>,&#x201D; <source>IEEE Journal on Selected Areas in Communications</source>, vol. <volume>38</volume>, no. <issue>12</issue>, pp. <fpage>2773</fpage>&#x2013;<lpage>2782</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Sanguinetti</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Zappone</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Debbah</surname></string-name></person-group>, &#x201C;<article-title>Deep learning power allocation in massive MIMO</article-title>,&#x201D; in <conf-name>2018 52nd Asilomar Conf. on Signals, Systems, and Computers</conf-name>, <conf-loc>USA</conf-loc>, pp. <fpage>1257</fpage>&#x2013;<lpage>1261</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Zappone</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Debbah</surname></string-name> and <string-name><given-names>Z.</given-names> <surname>Altman</surname></string-name></person-group>, &#x201C;<article-title>Online energy-efficient power control in wireless networks by deep neural networks</article-title>,&#x201D; in <conf-name>IEEE 19th Int. Workshop on Signal Processing Advances in Wireless Communications (SPAWC)</conf-name>, <conf-loc>Greece</conf-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>5</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Gao</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Lv</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Lu</surname></string-name></person-group>, &#x201C;<article-title>Deep Q-learning based dynamic resource allocation for self-powered ultra-dense networks</article-title>, in <conf-name>IEEE Int. Conf. on Communications Workshops (ICC Workshops)</conf-name>, <conf-loc>USA</conf-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>6</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Ali</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Haider</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Rahman</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Sohail</surname></string-name> and <string-name><given-names>Y. B.</given-names> <surname>Zikria</surname></string-name></person-group>, &#x201C;<article-title>Deep learning (DL) based joint resource allocation and RRH association in 5G-multi-tier networks</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>9</volume>, pp. <fpage>118357</fpage>&#x2013;<lpage>118366</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Ye</surname></string-name>, <string-name><given-names>G. Y.</given-names> <surname>Li</surname></string-name> and <string-name><given-names>B. F.</given-names> <surname>Juang</surname></string-name></person-group>, &#x201C;<article-title>Deep reinforcement learning based resource allocation for V2V communications</article-title>,&#x201D; <source>IEEE Transactions on Vehicular Technology</source>, vol. <volume>68</volume>, no. <issue>4</issue>, pp. <fpage>3163</fpage>&#x2013;<lpage>3173</lpage>, <year>April 2019</year>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Ron</surname></string-name> and <string-name><given-names>J. -R.</given-names> <surname>Lee</surname></string-name></person-group>, &#x201C;<article-title>DRL-based sum-rate maximization in D2D communication underlaid uplink cellular networks</article-title>,&#x201D; <source>IEEE Transactions on Vehicular Technology</source>, vol. <volume>70</volume>, no. <issue>10</issue>, pp. <fpage>11121</fpage>&#x2013;<lpage>11126</lpage>, <year>Oct. 2021</year>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Amiri</surname></string-name>, <string-name><given-names>M. A.</given-names> <surname>Almasi</surname></string-name>, <string-name><given-names>J. G.</given-names> <surname>Andrews</surname></string-name> and <string-name><given-names>H.</given-names> <surname>Mehrpouyan</surname></string-name></person-group>, &#x201C;<article-title>Reinforcement learning for self organization and power control of two-tier heterogeneous networks</article-title>,&#x201D; <source>IEEE Transactions on Wireless Communications</source>, vol. <volume>18</volume>, no. <issue>8</issue>, pp. <fpage>3933</fpage>&#x2013;<lpage>3947</lpage>, <year>Aug. 2019</year>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Cui</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Liu</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Nallanathan</surname></string-name></person-group>, &#x201C;<article-title>Multi-agent reinforcement learning-based resource allocation for UAV networks</article-title>,&#x201D; <source>IEEE Transactions on Wireless Communications</source>, vol. <volume>19</volume>, no. <issue>2</issue>, pp. <fpage>729</fpage>&#x2013;<lpage>743</lpage>, <year>Feb. 2020</year>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>X.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Jiao</surname></string-name> and <string-name><given-names>G.</given-names> <surname>Min</surname></string-name></person-group>, &#x201C;<article-title>Deep Q-Network based resource allocation for UAV-assisted Ultra-Dense Networks</article-title>,&#x201D; <source>Computer Networks</source>, vol. <volume>196</volume>, pp. <fpage>1</fpage>&#x2013;<lpage>10</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Xu</surname></string-name> and <string-name><given-names>X.</given-names> <surname>Zhou</surname></string-name></person-group>, &#x201C;<article-title>Resource allocation in unmanned aerial vehicle (UAV)-assisted wireless-powered internet of things</article-title>,&#x201D; <source>Sensors</source>, vol. <volume>19</volume>, no. <issue>8</issue>, pp. <fpage>1908</fpage>&#x2013;<lpage>1928</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>P.</given-names> <surname>Luong</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Gagnon</surname></string-name>, <string-name><given-names>L. -N.</given-names> <surname>Tran</surname></string-name> and <string-name><given-names>F.</given-names> <surname>Labeau</surname></string-name></person-group>, &#x201C;<article-title>Deep reinforcement learning-based resource allocation in cooperative UAV-assisted wireless networks</article-title>,&#x201D; <source>IEEE Transactions on Wireless Communications</source>, vol. <volume>20</volume>, no. <issue>11</issue>, pp. <fpage>7610</fpage>&#x2013;<lpage>7625</lpage>, <year>Nov. 2021</year>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Min</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Cao</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Chen</surname></string-name></person-group>, &#x201C;<article-title>Reinforcement learning-based UAVs resource allocation for integrated sensing and communication (ISAC) system</article-title>,&#x201D; <source>Electronics</source>, vol. <volume>11</volume>, no. <issue>3</issue>, pp. <fpage>441</fpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>Y. Y.</given-names> <surname>Munaye</surname></string-name>, <string-name><given-names>R. -T.</given-names> <surname>Juang</surname></string-name>, <string-name><given-names>H. -P.</given-names> <surname>Lin</surname></string-name> and <string-name><given-names>G. B.</given-names> <surname>Tarekegn</surname></string-name></person-group>, &#x201C;<article-title>Resource allocation for multi-UAV assisted IoT networks: A deep reinforcement learning approach</article-title>,&#x201D; in <conf-name>Int. Conf. on Pervasive Artificial Intelligence (ICPAI)</conf-name>, <conf-loc>Taiwan</conf-loc>, pp. <fpage>15</fpage>&#x2013;<lpage>22</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>van Hasselt</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Guez</surname></string-name> and <string-name><given-names>D.</given-names> <surname>Silver</surname></string-name></person-group>, &#x201C;<article-title>Deep reinforcement learning with double Q-learning</article-title>,&#x201D; in <conf-name>Proc. AAAI Conf. Artif. Intell.</conf-name>, <conf-loc>Phoenix, Arizona, USA</conf-loc>, pp. <fpage>2094</fpage>&#x2013;<lpage>2100</lpage>, <year>Sep. 2016</year>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H. A.</given-names> <surname>Shah</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Zhao</surname></string-name> and <string-name><given-names>I. -M.</given-names> <surname>Kim</surname></string-name></person-group>, &#x201C;<article-title>Joint network control and resource allocation for space-terrestrial integrated network through hierarchal deep actor-critic reinforcement learning</article-title>,&#x201D; <source>IEEE Transactions on Vehicular Technology</source>, vol. <volume>70</volume>, no. <issue>5</issue>, pp. <fpage>4943</fpage>&#x2013;<lpage>4954</lpage>, <year>May 2021</year>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. M.</given-names> <surname>Sande</surname></string-name>, <string-name><given-names>M. C.</given-names> <surname>Hlophe</surname></string-name> and <string-name><given-names>B. T.</given-names> <surname>Maharaj</surname></string-name></person-group>, &#x201C;<article-title>Access and radio resource management for IAB networks using deep reinforcement learning</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>9</volume>, pp. <fpage>114218</fpage>&#x2013;<lpage>114234</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Agiwal</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Roy</surname></string-name> and <string-name><given-names>N.</given-names> <surname>Saxena</surname></string-name></person-group>, &#x201C;<article-title>Next generation 5G wireless networks: A comprehensive survey</article-title>,&#x201D; <source>IEEE Communications Surveys &#x0026; Tutorials</source>, vol. <volume>18</volume>, no. <issue>3</issue>, pp. <fpage>1617</fpage>&#x2013;<lpage>1655</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Liang</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Ye</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Yu</surname></string-name> and <string-name><given-names>G. Y.</given-names> <surname>Li</surname></string-name></person-group>, &#x201C;<article-title>Deep-learning-based wireless resource allocation with application to vehicular networks</article-title>,&#x201D; in <source>Proc. of the IEEE</source>, vol. <volume>108</volume>, no. <issue>2</issue>, pp. <fpage>341</fpage>&#x2013;<lpage>356</lpage>, <year>Feb. 2020</year>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Iqbal</surname></string-name>, <string-name><given-names>M. -L.</given-names> <surname>Tham</surname></string-name> and <string-name><given-names>Y. C.</given-names> <surname>Chang</surname></string-name></person-group>, &#x201C;<article-title>Double deep Q-network-based energy-efficient resource allocation in cloud radio access network</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>9</volume>, pp. <fpage>20440</fpage>&#x2013;<lpage>20449</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Wang</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Xu</surname></string-name></person-group>, &#x201C;<article-title>Energy-efficient resource allocation in uplink NOMA systems with deep reinforcement learning</article-title>,&#x201D; in <conf-name>11th Int. Conf. on Wireless Communications and Signal Processing (WCSP)</conf-name>, <conf-loc>Xi&#x2019;an, China</conf-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>6</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>X.</given-names> <surname>Lai</surname></string-name>, <string-name><given-names>Q.</given-names> <surname>Hu</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Fei</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Huang</surname></string-name></person-group>, &#x201C;<article-title>Adaptive resource allocation method based on deep Q network for industrial internet of things</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>8</volume>, pp. <fpage>27426</fpage>&#x2013;<lpage>27434</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Hussain</surname></string-name>, <string-name><given-names>S. A.</given-names> <surname>Hassan</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Hussain</surname></string-name> and <string-name><given-names>E.</given-names> <surname>Hossain</surname></string-name></person-group>, &#x201C;<article-title>Machine learning for resource management in cellular and IoT networks: Potentials current solutions, and open challenges</article-title>,&#x201D; <source>IEEE Communications Surveys &#x0026; Tutorials</source>, vol. <volume>22</volume>, no. <issue>2</issue>, pp. <fpage>1251</fpage>&#x2013;<lpage>1275</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Yu</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Zhou</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Gong</surname></string-name> and <string-name><given-names>D.</given-names> <surname>Wu</surname></string-name></person-group>, &#x201C;<article-title>When deep reinforcement learning meets federated learning: Intelligent multitimescale resource management for multiaccess edge computing in 5G ultradense network</article-title>,&#x201D; <source>IEEE Internet of Things Journal</source>, vol. <volume>8</volume>, no. <issue>4</issue>, pp. <fpage>2238</fpage>&#x2013;<lpage>2251</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Zeng</surname></string-name> and <string-name><given-names>R.</given-names> <surname>Zhang</surname></string-name></person-group>, &#x201C;<article-title>Multi-antenna UAV data harvesting: Joint trajectory and communication optimization</article-title>,&#x201D; <source>Journal of Communications and Information Networks</source>, vol. <volume>5</volume>, no. <issue>1</issue>, pp. <fpage>86</fpage>&#x2013;<lpage>99</lpage>, <year>March 2020</year>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Pratap</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Misra</surname></string-name> and <string-name><given-names>S. K.</given-names> <surname>Das</surname></string-name></person-group>, &#x201C;<article-title>Maximizing fairness for resource allocation in heterogeneous 5G networks</article-title>,&#x201D; <source>IEEE Transactions on Mobile Computing</source>, vol. <volume>20</volume>, no. <issue>2</issue>, pp. <fpage>603</fpage>&#x2013;<lpage>619</lpage>, Feb. <year>2021</year>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Fu</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Yang</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Yang</surname></string-name>, <string-name><given-names>A. K. Y.</given-names> <surname>Wong</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Shi</surname></string-name>, <etal>et al.</etal>,</person-group>, &#x201C;<article-title>Energy-efficient offloading and resource allocation for mobile edge computing enabled mission-critical internet-of-things systems</article-title>,&#x201D; <source>EURASIP Journal on Wireless Communications and Networking</source>, vol. <volume>2021</volume>, no. <issue>1</issue>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Al-Hourani</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Kandeepan</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Lardner</surname></string-name></person-group>, &#x201C;<article-title>Optimal LAP altitude for maximum coverage</article-title>,&#x201D; <source>IEEE Wireless Commun. Lett.</source>, vol. <volume>3</volume>, no. <issue>6</issue>, pp. <fpage>569</fpage>&#x2013;<lpage>572</lpage>, <year>Dec. 2014</year>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. A.</given-names> <surname>Ouamri</surname></string-name>, <string-name><given-names>M. E.</given-names> <surname>Ote&#x015F;teanu</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Barb</surname></string-name> and <string-name><given-names>C.</given-names> <surname>Gueguen</surname></string-name></person-group> &#x201C;<article-title>Coverage analysis and efficient placement of drone-BSs in 5G networks</article-title>,&#x201D; <source>Engineering Proceedings</source>, vol. <volume>14</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>8</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Md. A.</given-names> <surname>Ouamri</surname></string-name></person-group> &#x201C;<article-title>Stochastic geometry modeling and analysis of downlink coverage and rate in small cell network</article-title>,&#x201D; <source>Telecommun Syst</source>, vol. <volume>77</volume>, no. <issue>4</issue>, pp. <fpage>767</fpage>&#x2013;<lpage>779</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Alkama</surname></string-name>, <string-name><given-names>M. A.</given-names> <surname>Ouamri</surname></string-name>, <string-name><given-names>M. S.</given-names> <surname>Alzaidi</surname></string-name>, <string-name><given-names>R. N.</given-names> <surname>Shaw</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Azni</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>Downlink performance analysis in MIMO UAV-cellular communication with LoS/NLoS propagation under 3D beamforming</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>10</volume>, pp. <fpage>6650</fpage>&#x2013;<lpage>6659</lpage>, <year>2022</year>.</mixed-citation></ref>
</ref-list>
</back>
</article>














