<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">28411</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2022.028411</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Air Pollution Prediction Via Graph Attention Network and Gated Recurrent Unit</article-title>
<alt-title alt-title-type="left-running-head">Air Pollution Prediction Via Graph Attention Network and Gated Recurrent Unit</alt-title>
<alt-title alt-title-type="right-running-head">Air Pollution Prediction Via Graph Attention Network and Gated Recurrent Unit</alt-title>
</title-group>
<contrib-group content-type="authors">
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Wang</surname><given-names>Shun</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Qiao</surname><given-names>Lin</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Fang</surname><given-names>Wei</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Jing</surname><given-names>Guodong</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Sheng</surname><given-names>Victor S.</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-6" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Zhang</surname><given-names>Yong</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><email>zhangyong2010@bjut.edu.cn</email>
</contrib>
<aff id="aff-1"><label>1</label><institution>Beijing Key Laboratory of Multimedia and Intelligent Software Technology, Beijing Artificial Intelligence Institute, the Faculty of Information Technology, Beijing University of Technology</institution>, <addr-line>Beijing, 100124</addr-line>, <country>China</country></aff>
<aff id="aff-2"><label>2</label><institution>Beijing Meteorological Observatory</institution>, <addr-line>Beijing, 100089</addr-line>, <country>China</country></aff>
<aff id="aff-3"><label>3</label><institution>Nanjing University of Information Science &#x0026; Technology</institution>, <addr-line>Nanjing, 210044</addr-line>, <country>China</country></aff>
<aff id="aff-4"><label>4</label><institution>China Meteorological Administration Training Centre</institution>, <addr-line>Beijing, 100081</addr-line>, <country>China</country></aff>
<aff id="aff-5"><label>5</label><institution>Texas Tech University</institution>, <addr-line>Lubbock, TX79409</addr-line>, <country>United States</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Yong Zhang. Email: <email>zhangyong2010@bjut.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub" date-type="pub" iso-8601-date="2022-05-16"><day>16</day>
<month>05</month>
<year>2022</year></pub-date>
<volume>73</volume>
<issue>1</issue>
<fpage>673</fpage>
<lpage>687</lpage>
<history>
<date date-type="received"><day>09</day><month>2</month><year>2022</year></date>
<date date-type="accepted"><day>23</day><month>3</month><year>2022</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2022 Wang et al.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Wang et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_28411.pdf"></self-uri>
<abstract>
<p>PM2.5 concentration prediction is of great significance to environmental protection and human health. Achieving accurate prediction of PM2.5 concentration has become an important research task. However, PM2.5 pollutants can spread in the earth&#x2019;s atmosphere, causing mutual influence between different cities. To effectively capture the air pollution relationship between cities, this paper proposes a novel spatiotemporal model combining graph attention neural network (GAT) and gated recurrent unit (GRU), named GAT-GRU for PM2.5 concentration prediction. Specifically, GAT is used to learn the spatial dependence of PM2.5 concentration data in different cities, and GRU is to extract the temporal dependence of the long-term data series. The proposed model integrates the learned spatio-temporal dependencies to capture long-term complex spatio-temporal features. Considering that air pollution is related to the meteorological conditions of the city, the knowledge acquired from meteorological data is used in the model to enhance PM2.5 prediction performance. The input of the GAT-GRU model consists of PM2.5 concentration data and meteorological data. In order to verify the effectiveness of the proposed GAT-GRU prediction model, this paper designs experiments on real-world datasets compared with other baselines. Experimental results prove that our model achieves excellent performance in PM2.5 concentration prediction.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Air pollution prediction</kwd>
<kwd>deep learning</kwd>
<kwd>spatiotemporal data modeling</kwd>
<kwd>graph attention network</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1"><label>1</label><title>Introduction</title>
<p>With the development of urban economy, air pollution has become more serious in recent years. This situation has received significant public attention. Major air pollutants include SO2, NO2, PM2.5, and PM10. PM2.5 (particulate matter with diameters less than or equal to 2.5 <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mrow><mml:mi>&#x03BC;</mml:mi><mml:mi mathvariant="normal">m</mml:mi></mml:mrow></mml:math></inline-formula>) have received great attention as a typical air pollutant. Many studies have proved that a high concentration of PM2.5 can harm people&#x2019;s health, such as damage to the respiratory and cardiovascular systems [<xref ref-type="bibr" rid="ref-1">1</xref>]. The average life expectancy of human beings is reduced due to long-term living in an environment with high air pollution [<xref ref-type="bibr" rid="ref-2">2</xref>]. People living in areas with high air pollution levels may suffer more from brain atrophy in Alzheimer&#x2019;s when they are old [<xref ref-type="bibr" rid="ref-3">3</xref>]. Therefore, accurate prediction of PM2.5 concentration can help the public take effective countermeasures to protect public health, and it can also help decision-makers of government formulate related environmental protection policies.</p>
<p>Air pollution data collected from monitoring stations in different cities has complex temporal and spatial characteristics. The monitoring data we obtained is composed of long-term series of PM2.5 concentration in multiple cities. These time series have two temporal characteristics: the tendency to increase or decrease over time and the seasonality in which air pollution becomes severe in certain seasons, such as winter. In addition to temporal characteristics, PM2.5 pollutants spread and influence each other between adjacent cities, so the spatial correlations between cities need to be considered in the prediction process. Existing studies usually do not consider spatial correlation [<xref ref-type="bibr" rid="ref-4">4</xref>&#x2013;<xref ref-type="bibr" rid="ref-6">6</xref>], or only consider fixed spatial correlation and cannot dynamically learn spatial features [<xref ref-type="bibr" rid="ref-7">7</xref>&#x2013;<xref ref-type="bibr" rid="ref-9">9</xref>]. On the other hand, air pollutant concentrations are affected by urban meteorological conditions, such as the city&#x2019;s humidity, temperature, precipitation, and wind speed. These meteorological conditions are underutilized in existing forecasting models. To address the above two limitations, we designed a new PM2.5 concentration prediction model GAT-GRU. The proposed model is able to learn the dynamic spatiotemporal dependence of air pollution data and make good use of the city&#x2019;s meteorological knowledge.</p>
<p>In order to achieve effective capture of complex spatial features, our paper attempts to use graph attention networks to learn spatial characteristics of PM2.5 concentration data. The graph attention network (GAT) obtains the feature representation of the target node by assigning different importance to different nodes in the neighborhood of the target node [<xref ref-type="bibr" rid="ref-10">10</xref>]. In the PM2.5 concentration prediction process, some neighboring cities have strong correlations with the target city in terms of air pollutants. Therefore, the GAT model can focus on the important cities with strong correlations to obtain a more accurate representation when learning the spatial dependence. In dealing with the complex temporal dependence of PM2.5 data sequences, we use another variant of the recurrent neural network: Gated Recurrent Unit (GRU) [<xref ref-type="bibr" rid="ref-11">11</xref>]. Compared with traditional recurrent neural networks, GRU can overcome the problems of gradient disappearance and gradient explosion when modeling long-range dependence. On the other hand, GRU has the advantage of rapid calculation speed due to fewer calculation parameters. In general, this paper combines GAT and GRU to form a spatiotemporal prediction model of PM2.5 pollutant concentration.</p>
<p>In this paper, we propose a hybrid model called GAT-GRU that integrates GAT module and GRU cell for spatiotemporal modeling and prediction of PM2.5 concentration. In addition, the GAT-GRU prediction model makes an attempt to incorporate meteorological knowledge into the graph structure as node attributes. In summary, GAT-GRU is a prediction model that can effectively capture the spatiotemporal dependence of PM2.5 concentration and use various additional information to enhance prediction.</p>
<p>The three main contributions of this paper are as follows:
<list list-type="simple">
<list-item><label>(1)</label><p>We propose a spatiotemporal prediction model called GAT-GRU. Graph Attention Networks are introduced in the model to learn spatial connections between nodes. This model can effectively learn the spatiotemporal dependence of PM2.5 concentration data series.</p></list-item>
<list-item><label>(2)</label><p>Meteorological knowledge that reflects the characteristics of the monitoring station itself is utilized in the predictive model. We incorporate meteorological knowledge as part of the input to the graph attention network.</p></list-item>
<list-item><label>(3)</label><p>The proposed model has been experimented on real-world datasets. The results validate the good performance of the model in PM2.5 prediction.</p></list-item>
</list></p>
</sec>
<sec id="s2"><label>2</label><title>Related Work</title>
<sec id="s2_1"><label>2.1</label><title>PM2.5 Concentration Prediction</title>
<p>Weather prediction is an important research direction [<xref ref-type="bibr" rid="ref-12">12</xref>&#x2013;<xref ref-type="bibr" rid="ref-15">15</xref>]. The main tasks include rainfall prediction, temperature prediction, air pollution prediction, etc. Recent research on PM2.5 concentration prediction is generally based on deep learning methods, which convert the PM2.5 concentration prediction problem into a data mining problem. Therefore, it is necessary to introduce the PM2.5 prediction models using deep learning methods. With the rapid growth of air pollution data, deep learning methods have been further applied in PM2.5 prediction and proven effective prediction performance. In order to capture the complex temporal characteristics contained in the air pollutant data series, recurrent neural networks such as Long Short-Term Memory (LSTM) [<xref ref-type="bibr" rid="ref-4">4</xref>] have been widely used in PM2.5 prediction and achieved good performance [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-6">6</xref>]. These studies show that LSTM can achieve better results than traditional machine learning methods when modeling long-term sequence prediction problems. However, these methods only consider the temporal characteristics of PM2.5 concentration series, and lack the utilization of spatial characteristics that reflect the correlation between different monitoring stations.</p>
<p>However, the aforementioned deep learning methods usually only consider temporal characteristics of PM2.5 concentration data. In the real world, the PM2.5 concentration data of different regions are spatially interrelated, and PM2.5 pollutants between areas could be transmitted and diffused to each other. In order to learn the spatial correlation, convolutional neural networks are introduced to extract the spatial characteristics of the time series of PM2.5 pollutants [<xref ref-type="bibr" rid="ref-16">16</xref>,<xref ref-type="bibr" rid="ref-17">17</xref>]. Many research works combine convolutional neural networks (CNN) and LSTM to learn the temporal and spatial dependence of urban PM2.5 concentration [<xref ref-type="bibr" rid="ref-7">7</xref>&#x2013;<xref ref-type="bibr" rid="ref-9">9</xref>]. Attention ConvLSTM Encoder-Forecaster (AttEF) [<xref ref-type="bibr" rid="ref-18">18</xref>] integrates the attention mechanism into ConvLSTM encoder-forecaster to solve the loss of important spatiotemporal information, which has achieved good performance in precipitation nowcasting. These methods combine CNN and LSTM to form a spatiotemporal prediction model for PM2.5 concentration. But convolutional neural networks can only be used to process data in Euclidean space, and there are still shortcomings in capturing spatial features. In general, deep learning methods have achieved good results in PM2.5 concentration prediction. How to learn the spatial dependence of PM2.5 concentration data between different monitored cities needs further research. In addition, the influence of meteorological factors needs to be considered in the forecasting process.</p>
</sec>
<sec id="s2_2"><label>2.2</label><title>Graph Neural Networks</title>
<p>Recently, graph neural networks have received increasing attention from researchers due to their ability to learn graph structure information, representing complex non-Euclidean spatial information [<xref ref-type="bibr" rid="ref-19">19</xref>]. Considering the non-Euclidean distribution among air monitoring stations in different cities, only using convolutional neural networks is not enough to capture complex spatial information. Therefore, the graph neural network (GNN) model based on the graph structure can better learn the spatial correlation between PM2.5 monitoring concentration data in different cities. PM2.5-GNN integrates domain knowledge into graph-structured data to explicitly model the long-term spatiotemporal dependence in the PM2.5 forecasting process [<xref ref-type="bibr" rid="ref-20">20</xref>]. In addition to GNN, graph convolutional neural networks also play an essential role in air pollutant prediction. In the GLSTM model [<xref ref-type="bibr" rid="ref-21">21</xref>], the graph convolutional network is combined with LSTM to introduce spatiotemporal information into PM2.5 concentration prediction. Hierarchical graph convolutional networks are adopted to model air pollutants&#x2019; diffusion process more effectively in air quality prediction [<xref ref-type="bibr" rid="ref-22">22</xref>]. GCLSTM proposes a hybrid model combining graph convolutional network and LSTM to model and predict the continuous changes of PM2.5 concentration [<xref ref-type="bibr" rid="ref-23">23</xref>]. The above methods use graph neural networks to learn node features on a fixed graph, and cannot dynamically learn the weights of edges representing correlations between nodes. Graph convolutional networks or graph neural networks obtain node representations by aggregating the proximity information of target nodes. However, the relationship between different PM2.5 monitoring sites is not just a connection of 0 or 1. The spread of air pollution between cities is also closely related to meteorological conditions. The connection relationship between nodes needs to be more optimized to a more accurate value to obtain a richer expression. Therefore, we use graph attention network to learn the spatial relationship of PM2.5 concentration data in this paper. Compared with other types of graph networks, graph attention networks use the attention mechanism to learn the relative importance of different neighbor nodes. This method can effectively improve the expressive ability of the graph network.</p>
</sec>
</sec>
<sec id="s3"><label>3</label><title>Data and Meteorological Knowledge</title>
<p>The dataset used in the paper is a public dataset in the field of air pollution research [<xref ref-type="bibr" rid="ref-20">20</xref>]. The dataset contains PM2.5 concentration and meteorological feature data in 184 cities across multiple provinces in north and south China. The time of the collected dataset is from January 1, 2015 to December 31, 2018, which is recorded every three hours. Following the previous work [<xref ref-type="bibr" rid="ref-20">20</xref>], this public dataset can be divided into two datasets. Dataset 1 uses the pollution situation in the past period to predict the future PM2.5 concentrations. Dataset 2 selects the monitoring data during winter, when the air pollution is more serious, and the pollutants are blown by the monsoon from northern China to southern China. <xref ref-type="fig" rid="fig-1">Fig. 1</xref> shows the specific locations of the 184 cities in the dataset, (a) and (b) are the locations of cities with PM2.5 monitoring data in northern China and southern China.</p>
<fig id="fig-1"><label>Figure 1</label><caption><title>The location of cities with PM2.5 monitoring data on the map</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28411-fig-1.png"/></fig>
<p>In the spatiotemporal modeling problem of PM2.5 prediction, we define the graph representing the spatial correlation between cities as <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mi>G</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>E</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. <italic>V</italic> represents the city node, and <italic>E</italic> represents the correlation between the nodes. We need to construct an adjacency matrix representing the graph structure according to the distance between cities. When the spatial distance between two cities is within a specific range and there are no high-altitude mountains between them, the two cities can be judged as having a strong PM2.5 concentration correlation. The construction method of the adjacency matrix in this article is as follows:
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mspace width="thickmathspace" /><mml:mn>1</mml:mn><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x003C;</mml:mo><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x003C;</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mspace width="thickmathspace" /></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mn>0</mml:mn><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x2265;</mml:mo><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x2265;</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> represents the distance between two cities, <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> represents the highest elevation of the mountains between the two cities. In this paper, <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow><mml:mspace width="thickmathspace" /></mml:math></inline-formula> is set to 300&#x2005;km and <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is set to 1200&#x2005;m.</p>
<p><bold>Meteorological Knowledge</bold>: The meteorological characteristics and environmental factors of cities largely affect the production or spread of PM2.5 pollutants. SCENT [<xref ref-type="bibr" rid="ref-24">24</xref>] proposes that the precipitation results in precipitation nowcasting are related to non-image features such as wind speed and shape of cloud clusters. The study find that there is a negative correlation between temperature and PM2.5 concentration. As the temperature increases, the particle concentration decreases. Air pressure is positively related to particle concentration. There is a negative correlation between wind speed and PM2.5 concentration within a certain range [<xref ref-type="bibr" rid="ref-25">25</xref>]. Therefore, we also integrate meteorological features as domain knowledge into the process of air pollutant prediction. The domain knowledge of meteorological characteristics related to PM2.5 concentration includes Planetary Boundary Layer (PBL) height, stability index of tropospheric stratification, wind speed, temperature, high surface relative humidity, precipitation and surface pressure. <xref ref-type="table" rid="table-1">Tab. 1</xref> shows the names and units of seven types of meteorological knowledge. In the GAT-GRU model, meteorological knowledge is utilized as attributes of different city nodes to enhance PM2.5 concentration prediction.</p>

<table-wrap id="table-1"><label>Table 1</label><caption><title>Meteorological knowledge of cities</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Name</th>
<th align="left">Unit</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Planetary Boundary Layer (PBL) height</td>
<td align="left"><italic>m</italic></td>
</tr>
<tr>
<td align="left">stability index of tropospheric stratification</td>
<td align="left"><italic>K</italic></td>
</tr>
<tr>
<td align="left">wind speed</td>
<td align="left"><italic>m/s</italic></td>
</tr>
<tr>
<td align="left">temperature</td>
<td align="left"><italic>K</italic></td>
</tr>
<tr>
<td align="left">high surface relative humidity</td>
<td align="left"><italic>&#x0025;</italic></td>
</tr>
<tr>
<td align="left">precipitation</td>
<td align="left"><italic>m</italic></td>
</tr>
<tr>
<td align="left">surface pressure</td>
<td align="left"><italic>Pa</italic></td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4"><label>4</label><title>The Proposed Method</title>
<p>PM2.5 concentration prediction can be regarded as a spatiotemporal modeling problem. This paper uses two deep learning methods to learn the spatiotemporal dependence of PM2.5 concentration data. This paper uses two deep learning methods to construct a GAT-GRU model to learn the temporal and spatial dependence of PM2.5 concentration data. Graph attention network is used for spatial feature modeling, and the gated recurrent unit is used for temporal feature modeling.</p>
<sec id="s4_1"><label>4.1</label><title>Spatial Feature Modeling</title>
<p>For the air pollutant prediction problem, it is vital to learn the spatial characteristics and dependencies contained in the original data. From a spatial perspective, neighboring cities generally have similar air pollution conditions, and air pollutants could spread and affect each other between neighboring cities. The current research work either ignores the mutual influence between different city nodes or introduces prior knowledge to establish node correlations. In GAT-GRU prediction model, the graph attention network is used to capture the spatial dependence of PM2.5 concentration monitoring data. Compared with graph convolutional network, GAT can assign different weights to the neighbor nodes of the target node according to their importance.</p>
<p>Unlike the general GAT-based forecasting model, the input <italic>h</italic> of the GAT layer in the GAT-GRU model is obtained by combining two parts: PM2.5 concentration data <italic>x</italic> for a period of time, and the domain knowledge <italic>s</italic> reflecting the city&#x2019;s meteorological conditions during this period. Meteorological conditions are closely related to the generation and spread of air pollutions. Therefore, these factors need to be fully considered in the PM2.5 concentration prediction process.</p>
<p>The GAT layer in the prediction model is mainly composed of two parts: (1) calculate the attention coefficient. (2) aggregate features of neighbor nodes to get node representation. <xref ref-type="fig" rid="fig-3">Fig. 3</xref> shows the calculation process of the graph attention mechanism. First of all, the attention coefficient represents the importance of neighboring nodes to the target node. The following formula can calculate the attention coefficient:
<fig id="fig-2"><label>Figure 2</label><caption><title>Graph attention mechanism</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28411-fig-2.png"/></fig>
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi>W</mml:mi><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>W</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p><inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> represents the attention coefficient between neighboring node <italic>i</italic> and target node <italic>j</italic>. The value of the attention coefficient reflects the strength of the relationship between the two nodes. To make the attention coefficient comparable between all nodes, the <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:math></inline-formula> function is used to normalize the attention coefficient, and the formula is as follows:
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x2061;</mml:mo><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mstyle><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable><mml:mspace width="thickmathspace" /></mml:math></disp-formula></p>
<p>The calculation formula of the final result of the attention coefficient is shown in formula <xref ref-type="disp-formula" rid="eqn-4">(4)</xref>. The specific attention operation in GAT is to splicing the feature vectors of two nodes together, and then doing an inner product with the weight vector <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:mrow><mml:msup><mml:mrow><mml:mover><mml:mi>a</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow><mml:mi>T</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula>.
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mrow><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>L</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>k</mml:mi><mml:mi>y</mml:mi><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>L</mml:mi><mml:mi>U</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mover><mml:mi>a</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>T</mml:mi></mml:msup></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mi>W</mml:mi><mml:mo>[</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>s</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mspace width="thickmathspace" /><mml:mi>W</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>s</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo stretchy="false">]</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:munder><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:munder><mml:mo>&#x2061;</mml:mo><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>L</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>k</mml:mi><mml:mi>y</mml:mi><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>L</mml:mi><mml:mi>U</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mover><mml:mi>a</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>T</mml:mi></mml:msup></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mi>W</mml:mi><mml:mo>[</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>s</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mspace width="thickmathspace" /><mml:mi>W</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>k</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>s</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo stretchy="false">]</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> represents all neighboring nodes of node <italic>i</italic>, <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mo fence="false" stretchy="false">|</mml:mo></mml:math></inline-formula> represents the concatenation operation, <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:mi>L</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>k</mml:mi><mml:mi>y</mml:mi><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>L</mml:mi><mml:mi>U</mml:mi></mml:math></inline-formula> denotes the nonlinear activation function. After calculating the weight of each city node&#x2019;s neighboring city nodes, the output of the GAT layer can be obtained by aggregating the information of the neighboring nodes.
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mi>i</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:munder><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:munder><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mi>W</mml:mi><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mi>&#x03C3;</mml:mi></mml:math></inline-formula> represents the activation function, <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mspace width="thickmathspace" /><mml:msubsup><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> is the node feature vector obtained calculated by the attention mechanism. In addition to a separate attention mechanism, multi-head attention can ensure the stability of the attention mechanism. Multi-head attention allows the model to have the ability to learn relevant information from different subspaces. K represents the number of attention layers in the multi-head attention mechanism.
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mn>1</mml:mn><mml:mi>K</mml:mi></mml:mfrac></mml:mstyle><mml:munderover><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>K</mml:mi></mml:munderover><mml:mspace width="thickmathspace" /><mml:munder><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:munder><mml:mspace width="thickmathspace" /><mml:msubsup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mi>W</mml:mi><mml:mi>k</mml:mi></mml:msup></mml:mrow><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>In order to enhance the stability of the results, the multi-head attention mechanism is used in the GAT layer. As shown in formula <xref ref-type="disp-formula" rid="eqn-6">(6)</xref>, the results of <italic>K</italic> independent attention operations are aggregated to obtain the final feature representation.</p>
</sec>
<sec id="s4_2"><label>4.2</label><title>Temporal Feature Modeling</title>
<p>The PM2.5 concentration data recorded by the air pollution monitoring stations is stored in the form of time series. The time series of PM2.5 concentrations have remarkable features such as periodicity, proximity and trend. Periodicity means that the PM2.5 concentration fluctuates cyclically over a longer period of time. Proximity means that the PM2.5 concentration values are closer when the time period is similar. Trend means that the change of PM2.5 concentration has a trend of increase or decrease in a period of time. Therefore, it is very important to model the temporal dependence of PM2.5 concentration data. With the development of deep learning, the recurrent neural network has become an effective method in time series modeling. Many PM2.5 prediction methods use LSTM as the basic model for learning temporal dependencies [<xref ref-type="bibr" rid="ref-7">7</xref>&#x2013;<xref ref-type="bibr" rid="ref-9">9</xref>]. This paper uses a variant of the recurrent neural network called gated recurrent unit (GRU) to process air pollution data. <xref ref-type="fig" rid="fig-3">Fig. 3</xref> shows the overall structure of the gated recurrent<?TeX \nobreak?> <?TeX \hbox\bgroup?>unit.<?TeX \egroup?></p>
<p>Gated recurrent unit contains two gates: reset gate <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and update gate <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. The reset gate <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> determines the combination of the new input information <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mspace width="thickmathspace" /></mml:math></inline-formula> and the previous memory state <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula>. The update gate <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> determines the amount of past state information <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mspace width="thickmathspace" /></mml:math></inline-formula> that continues to be saved in the current state <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:mrow><mml:mtext>&#xA0;</mml:mtext></mml:mrow><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. <xref ref-type="fig" rid="fig-3">Fig. 3</xref> shows the internal structure of the GRU and the connection between the update gate and the reset gate. The following is the calculation formula of GRU:
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mi>z</mml:mi></mml:msub></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<fig id="fig-3"><label>Figure 3</label><caption><title>The overall structure of the gated recurrent unit</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28411-fig-3.png"/></fig>
<p><inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> represent the output of reset gate and update gate, <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mi>z</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> represent learnable parameters. <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> represents the input data at the current time <italic>t</italic>. In the GAT-GRU prediction model, <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> includes PM2.5 concentration data and meteorological characteristic data.
<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>h</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mi>tanh</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>h</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p><inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> represents the output at time <italic>t</italic>, <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> represent the output of the previous time <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> and the input of this time <italic>t</italic>.</p>
</sec>
<sec id="s4_3"><label>4.3</label><title>GAT-GRU Model</title>
<p>In order to model the spatiotemporal dependence of PM2.5 concentration sequence, this paper proposes the GAT-GRU model composed of graph attention mechanism and gated recurrent unit. The input of the GAT-GRU model includes the node features matrix <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:mi>S</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, the PM2.5 concentration data <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and the adjacency matrix <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:mi>A</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>. <italic>N</italic> represents the number of cities with PM2.5 concentration monitoring data. The node features matrix <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:mrow><mml:msup><mml:mi>S</mml:mi><mml:mi>t</mml:mi></mml:msup></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> represents the meteorological knowledge. <xref ref-type="fig" rid="fig-4">Fig. 4</xref> shows the overall architecture of the GAT-GRU model.</p>
<fig id="fig-4"><label>Figure 4</label><caption><title>Spatial-temporal modeling using GAT-GRU cell</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28411-fig-4.png"/></fig>
<p>In this paper, the PM2.5 concentration data <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mspace width="thickmathspace" /></mml:math></inline-formula> and additional factors <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> at time <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:mi>t</mml:mi><mml:mspace width="thickmathspace" /></mml:math></inline-formula> are used as the input of the model, and the output predicted value <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:msubsup><mml:mi>X</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> of the GAT-GRU model and the additional factors <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> at the next moment are used as the input for the next step to continue the prediction. The basic unit of the GAT-GRU prediction model is the GAT-GRU cell. Each GAT-GRU cell mainly consists of three parts: the GAT layer, the GRU layer and fully connected layer. The proposed model predicts the future PM2.5 concentration value in a rolling manner. Formula <xref ref-type="disp-formula" rid="eqn-11">(11)</xref> to formula <xref ref-type="disp-formula" rid="eqn-17">(17)</xref> represent the calculation process of the GAT-GRU model.
<disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>X</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mi>c</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:mi>f</mml:mi><mml:mi>c</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represents the fully connected layer, which changes the dimension of input or output. <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:msubsup><mml:mi>X</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup><mml:mspace width="thickmathspace" /></mml:math></inline-formula> denotes the concatenation of the input concentration data <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and additional factors <inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. In this part, the PM2.5 concentration data and meteorological feature data at the current time t are processed as the input of the graph attention network. Meteorological feature data, as important factors related to PM2.5 concentration, provide important information for prediction models.
<disp-formula id="eqn-12"><label>(12)</label><mml:math id="mml-eqn-12" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>X</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mi>X</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msubsup><mml:mi>X</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p><inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mspace width="thickmathspace" /><mml:msubsup><mml:mi>X</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /></mml:math></inline-formula> represents the graph attention network layer. <italic>A</italic> is an adjacency matrix representing the spatial connection between different cities. Then the output of GAT layer is used as the input of GRU to obtain temporal dependence.
<disp-formula id="eqn-13"><label>(13)</label><mml:math id="mml-eqn-13" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mi>z</mml:mi></mml:msub></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>X</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-14"><label>(14)</label><mml:math id="mml-eqn-14" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>X</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-15"><label>(15)</label><mml:math id="mml-eqn-15" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>h</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mi>tanh</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>X</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-16"><label>(16)</label><mml:math id="mml-eqn-16" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>h</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>the above four formulas are the calculation formulas of GRU.<inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:mrow><mml:mtext>&#xA0;</mml:mtext></mml:mrow><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> is the output of the GAT-GRU cell at the last time step <inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:mspace width="thickmathspace" /><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>, which is used as the previous state&#x2019;s input at time <italic>t</italic>. <inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>X</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> is the input of GRU model.
<disp-formula id="eqn-17"><label>(17)</label><mml:math id="mml-eqn-17" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>X</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mi>c</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-50"><mml:math id="mml-ieqn-50"><mml:msubsup><mml:mi>X</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> denotes the predicted results of PM2.5 concentration in the next time <inline-formula id="ieqn-51"><mml:math id="mml-ieqn-51"><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>. In the GAT-GRU model, a rolling prediction model is used to predict the PM2.5 concentration after <italic>T</italic> time steps. <inline-formula id="ieqn-52"><mml:math id="mml-ieqn-52"><mml:msubsup><mml:mi>X</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula id="ieqn-53"><mml:math id="mml-ieqn-53"><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> are used together as the input of prediction cell for the next time<?TeX \nobreak?> <?TeX \hbox\bgroup?>step.<?TeX \egroup?></p>
<p>We summarize the learning process of the GAT-GRU prediction model in Algorithm 1 below.
</p>
<fig id="fig-5">
<graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28411-fig-5.png"/>
</fig>
</sec>
</sec>
<sec id="s5"><label>5</label><title>Experiments</title>
<sec id="s5_1"><label>5.1</label><title>Experiment Setting</title>
<p>We conduct experiments on a GPU server with a single 2080ti which has 11G video memory. We use PyTorch as the deep learning running framework of the server. The initial hidden state <inline-formula id="ieqn-70"><mml:math id="mml-ieqn-70"><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> of GRU is initialized with a zero tensor. The input data of all experiments is 1 step (3 h), and the output prediction result is 24 steps (72 h). Therefore, it means that 3 h of historical data is used in the experiment to implement the prediction of the PM2.5 concentration value after 72 h in the future. All models are trained for 100 epochs. The learning rate is 0.0005, and the batchsize is set to 64.</p>
<p>A total of 5 evaluation indicators are used in the experiment to evaluate the predictive performance of the GAT-GRU model. These indicators can be divided into two categories. The commonly used indicators to measure prediction accuracy in prediction models are mean absolute error (MAE) and root mean square error (RMSE). The other is the commonly used accuracy evaluation indicators in meteorology: critical success index (CSI), false alarm rate (FAR) and probability of detection (POD).</p>
<p>Specifically, the calculation methods of MAE and RMSE are as follows:
<disp-formula id="eqn-18"><label>(18)</label><mml:math id="mml-eqn-18" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mspace width="thickmathspace" /></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>E</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x03A9;</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mstyle><mml:munder><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>&#x03B5;</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x03A9;</mml:mi></mml:mrow></mml:mrow></mml:munder><mml:mo>&#x2061;</mml:mo><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-19"><label>(19)</label><mml:math id="mml-eqn-19" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mspace width="thickmathspace" /></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msqrt><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x03A9;</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mstyle><mml:munder><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>&#x03B5;</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x03A9;</mml:mi></mml:mrow></mml:mrow></mml:munder><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:msqrt><mml:mspace width="thickmathspace" /><mml:mo>,</mml:mo><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-71"><mml:math id="mml-ieqn-71"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula id="ieqn-72"><mml:math id="mml-ieqn-72"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> respectively represent the predicted value and the ground truth, <inline-formula id="ieqn-73"><mml:math id="mml-ieqn-73"><mml:mrow><mml:mi mathvariant="normal">&#x03A9;</mml:mi></mml:mrow></mml:math></inline-formula> denotes the total number of data samples.</p>
<p>Following PM2.5-GNN, the calculation methods of three meteorological evaluation indicators are as follows:
<disp-formula id="eqn-20"><label>(20)</label><mml:math id="mml-eqn-20" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mi>C</mml:mi><mml:mi>S</mml:mi><mml:mi>I</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:mi>f</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:mfrac></mml:mstyle><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-21"><label>(21)</label><mml:math id="mml-eqn-21" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mi>P</mml:mi><mml:mi>O</mml:mi><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:mfrac></mml:mstyle><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-22"><label>(22)</label><mml:math id="mml-eqn-22" display="block"><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mi>F</mml:mi><mml:mi>A</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:mi>f</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:mi>f</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:mfrac></mml:mstyle><mml:mspace width="thickmathspace" /><mml:mo>,</mml:mo><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /><mml:mspace width="thickmathspace" /></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-74"><mml:math id="mml-ieqn-74"><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi></mml:math></inline-formula> means the predicted value and the true value are both 1, and <inline-formula id="ieqn-75"><mml:math id="mml-ieqn-75"><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:math></inline-formula> indicates that the predicted value is 0 while the true value is 1, and <inline-formula id="ieqn-76"><mml:math id="mml-ieqn-76"><mml:mi>f</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi></mml:math></inline-formula> means that the predicted value is 1 and the true value is 0.</p>
</sec>
<sec id="s5_2"><label>5.2</label><title>Dataset and Baselines</title>
<p>As shown in Section 3, the dataset we used in the experiment is <bold>KnowAir</bold>, which contains the PM2.5 concentration data and meteorological attribute data of 184 cities in China collected from the real world from January 1, 2015 to December 31, 2018. From this dataset, we obtain two datasets (Dataset 1 and Dataset 2) for experiments. Dataset 1 represents the air pollution prediction under normal circumstances, and Dataset 2 selects the PM2.5 data in winter with severe air pollution for prediction. <xref ref-type="table" rid="table-2">Tab. 2</xref> shows the segmentation method of the two datasets in the experiment.</p>
<table-wrap id="table-2"><label>Table 2</label><caption><title>Segmentation of the dataset</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left"/>
<th align="left">Dataset 1</th>
<th align="left">Dataset 2</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Train</td>
<td align="left">2016/9/1&#x2013;2016/11/30</td>
<td align="left">2015/11/1&#x2013;2016/2/28</td>
</tr>
<tr>
<td align="left">Validate</td>
<td align="left">2016/12/1&#x2013;2016/12/31</td>
<td align="left">2016/11/1&#x2013;2017/2/28</td>
</tr>
<tr>
<td align="left">Test</td>
<td align="left">2017/1/1&#x2013;2017/1/31</td>
<td align="left">2017/11/1&#x2013;2018/2/28</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In the PM2.5 concentration prediction experiment, the following models are used as baselines compared with the proposed GAT-GRU model. For the fairness of the comparison of experimental results, we add meteorological knowledge as part of the model input when conducting experiments on all baselines.
<list list-type="simple">
<list-item><label>(1)</label><p><bold>MLP</bold> [<xref ref-type="bibr" rid="ref-26">26</xref>]<bold>:</bold> MLP is a classic multi-layer neural network model, which generally consists of an input layer, a hidden layer and an output layer. The representation of the node is used as the input of the multi-layer perceptron to obtain the prediction result finally.</p></list-item>
<list-item><label>(2)</label><p><bold>LSTM</bold> [<xref ref-type="bibr" rid="ref-4">4</xref>]<bold>:</bold> LSTM is an improved variant of recurrent neural network that can capture the time series characteristics of air pollution<?TeX \nobreak?> <?TeX \hbox\bgroup?>data.<?TeX \egroup?></p></list-item>
<list-item><label>(3)</label><p><bold>GRU</bold> [<xref ref-type="bibr" rid="ref-11">11</xref>]<bold>:</bold> GRU is another variant of the recurrent neural network. Similar to LSTM, GRU is also used to model the temporal characteristics of PM2.5 concentration data. The difference between GRU and the proposed model lies in the use of spatial feature modeling methods. Using GRU as a baseline can demonstrate the effectiveness of spatial modeling.</p></list-item>
<list-item><label>(4)</label><p><bold>GC-LSTM</bold> [<xref ref-type="bibr" rid="ref-23">23</xref>]<bold>:</bold> GC-LSTM is a spatiotemporal representation model with superior performance in the current research direction of PM2.5 prediction. This model combines GCN and LSTM to model the spatiotemporal characteristics of PM2.5 concentration<?TeX \nobreak?> <?TeX \hbox\bgroup?>data.<?TeX \egroup?></p></list-item>
<list-item><label>(5)</label><p><bold>PM2.5-GNN</bold> [<xref ref-type="bibr" rid="ref-20">20</xref>]<bold>:</bold> PM2.5-GNN is currently the state-of-the-art model for PM2.5 prediction performance. This model considers the use of the domain knowledge of city nodes to enhance the prediction effect and considers the attributes of the edges between cities, such as the transport effect brought by the<?TeX \nobreak?> <?TeX \hbox\bgroup?>wind.<?TeX \egroup?></p></list-item>
</list></p>
</sec>
<sec id="s5_3"><label>5.3</label><title>Results and Discussion</title>
<p><bold>Experiment 1: comparison with baselines</bold>. <xref ref-type="table" rid="table-3">Tabs. 3</xref> and <xref ref-type="table" rid="table-4">4</xref> show the PM2.5 concentration prediction performance of our proposed method and other methods used as baselines. As mentioned above, we conduct experiments on two real-world datasets. Experimental results include MAE, RMSE, CSI, POD and FAR. All the best experimental results are highlighted in bold. From the results in the table, we can see that the prediction performance of the methods that use the recurrent neural network to model the time characteristics is better than the MLP model. Furthermore, the predictive models that use graph structure to model spatial dependence have better performance, such as GC-LSTM, PM2.5-GNN and the proposed GAT-GRU model. These results mean that it is vital to model spatiotemporal dependence for PM2.5 prediction problem. Both GAT-GRU and PM2.5-GNN introduce new information and knowledge, such as meteorological attributes of cities and edge attributes obtained from wind speed and direction between city nodes. The experimental results prove that the introduction of meteorological knowledge can effectively improve the accuracy of prediction. In the experimental results of Dataset 1, the MAE and RMSE of the GAT-GRU model are 34.56 and 42.79, which are better than the results of other models. In the experimental results of Dataset 2, the results of GAT-GRU are also basically stronger than other models. Compared with the graph neural network in GC-LSTM and PM2.5-GNN, the graph attention network can effectively model the dynamic connection between monitoring nodes, especially under the condition of the integration of meteorological knowledge. In the experiments of the two datasets, the POD (Probability of Detection) indicator of PM2.5-GNN is better than the GAT-GRU model. Since PM2.5-GNN utilizes edge attributes composed of wind speed and wind direction between city nodes, more accurate PM2.5 propagation information can effectively enhance the probability of detection. In general, compared with other methods, the proposed GAT-GRU model achieves better prediction performance.</p>
<table-wrap id="table-3"><label>Table 3</label><caption><title>Overall performance on dataset 1. Best scores are in bold</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Method</th>
<th align="left">MAE</th>
<th align="left">RMSE</th>
<th align="left">CSI</th>
<th align="left">POD</th>
<th align="left">FAR</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">MLP</td>
<td align="left">41.89</td>
<td align="left">50.70</td>
<td align="left">52.44</td>
<td align="left">74.16</td>
<td align="left">35.25</td>
</tr>
<tr>
<td align="left">LSTM</td>
<td align="left">37.79</td>
<td align="left">46.19</td>
<td align="left">58.85</td>
<td align="left">81.03</td>
<td align="left">31.71</td>
</tr>
<tr>
<td align="left">GRU</td>
<td align="left">37.94</td>
<td align="left">46.06</td>
<td align="left">59.16</td>
<td align="left">83.32</td>
<td align="left">32.86</td>
</tr>
<tr>
<td align="left">GC-LSTM</td>
<td align="left">37.46</td>
<td align="left">45.71</td>
<td align="left">58.98</td>
<td align="left">81.92</td>
<td align="left">32.18</td>
</tr>
<tr>
<td align="left">PM2.5-GNN</td>
<td align="left">36.32</td>
<td align="left">44.36</td>
<td align="left">60.57</td>
<td align="left"><bold>83.94</bold></td>
<td align="left">31.37</td>
</tr>
<tr>
<td align="left">GAT-GRU</td>
<td align="left"><bold>34.56</bold></td>
<td align="left"><bold>42.79</bold></td>
<td align="left"><bold>61.71</bold></td>
<td align="left">81.95</td>
<td align="left"><bold>28.55</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="table-4"><label>Table 4</label><caption><title>Overall performance on dataset 2. Best scores are in bold</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Method</th>
<th align="left">MAE</th>
<th align="left">RMSE</th>
<th align="left">CSI</th>
<th align="left">POD</th>
<th align="left">FAR</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">MLP</td>
<td align="left">28.67</td>
<td align="left">35.55</td>
<td align="left">45.52</td>
<td align="left">60.85</td>
<td align="left">34.56</td>
</tr>
<tr>
<td align="left">LSTM</td>
<td align="left">26.90</td>
<td align="left">33.53</td>
<td align="left">49.75</td>
<td align="left">64.94</td>
<td align="left">31.88</td>
</tr>
<tr>
<td align="left">GRU</td>
<td align="left">26.54</td>
<td align="left">33.09</td>
<td align="left">49.83</td>
<td align="left">64.58</td>
<td align="left">31.31</td>
</tr>
<tr>
<td align="left">GC-LSTM</td>
<td align="left">26.57</td>
<td align="left">33.20</td>
<td align="left">50.13</td>
<td align="left">64.54</td>
<td align="left">30.73</td>
</tr>
<tr>
<td align="left">PM2.5-GNN</td>
<td align="left">25.68</td>
<td align="left">32.11</td>
<td align="left">51.35</td>
<td align="left"><bold>66.24</bold></td>
<td align="left">30.11</td>
</tr>
<tr>
<td align="left">GAT-GRU</td>
<td align="left"><bold>25.15</bold></td>
<td align="left"><bold>31.88</bold></td>
<td align="left"><bold>51.57</bold></td>
<td align="left">62.93</td>
<td align="left"><bold>26.56</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p><bold>Experiment 2: The influence of meteorological knowledge</bold>. <xref ref-type="table" rid="table-5">Tab. 5</xref> shows the results of ablation experiments on whether meteorological knowledge is incorporated in the GAT-GRU model. Taking the experimental results on Dataset 1 as an example, the predicted MAE and RMSE of the GAT-GRU model (without meteorological knowledge) are 42.21 and 50.61. With the use of meteorological knowledge, MAE and RMSE are reduced by 7.65 and 7.82 respectively. The experimental results show that the use of meteorological knowledge effectively improves the results of PM2.5 concentration prediction. In addition, the MAE and RMSE of the ablation experiment on Dataset 2 have a more significant decrease, which proves that the result of the use of meteorological knowledge on Dataset 2 is better than that of Dataset 1.</p>
<table-wrap id="table-5"><label>Table 5</label><caption><title>Results of ablation experiments using meteorological knowledge in prediction models</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Dataset</th>
<th align="left">Metric</th>
<th align="left">GAT-GRU</th>
<th align="left">GAT-GRU<break/>(no meteorological knowledge)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" rowspan="5">1</td>
<td align="left">RMSE</td>
<td align="left">42.79</td>
<td align="left">50.61</td>
</tr>
<tr>
<td align="left">MAE</td>
<td align="left">34.56</td>
<td align="left">42.21</td>
</tr>
<tr>
<td align="left">CSI</td>
<td align="left">61.71</td>
<td align="left">53.98</td>
</tr>
<tr>
<td align="left">POD</td>
<td align="left">81.95</td>
<td align="left">82.17</td>
</tr>
<tr>
<td align="left">FAR</td>
<td align="left">28.55</td>
<td align="left">38.83</td>
</tr>
<tr>
<td align="left" rowspan="5">2</td>
<td align="left">RMSE</td>
<td align="left">31.88</td>
<td align="left">39.45</td>
</tr>
<tr>
<td align="left">MAE</td>
<td align="left">25.15</td>
<td align="left">32.38</td>
</tr>
<tr>
<td align="left">CSI</td>
<td align="left">51.57</td>
<td align="left">38.55</td>
</tr>
<tr>
<td align="left">POD</td>
<td align="left">62.93</td>
<td align="left">55.56</td>
</tr>
<tr>
<td align="left">FAR</td>
<td align="left">26.56</td>
<td align="left">44.25</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><bold>Experiment 3: multi-head attention mechanism</bold>. <xref ref-type="table" rid="table-6">Tab. 6</xref> shows the experimental results of ablation for the multi-head attention mechanism. In the spatiotemporal prediction problem, we mainly consider two indicators, MAE and RMSE. As can be seen from the table, the best prediction results on Dataset 1 and Dataset 2 can be obtained when the numbers of multi-head attention mechanisms are 2 and 6. The results of Experiment 3 demonstrate the effectiveness of the multi-head attention mechanism.</p>
<table-wrap id="table-6"><label>Table 6</label><caption><title>Results of ablation experiments of heads number in GAT</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Dataset</th>
<th align="left">Multi-Heads number</th>
<th align="left">MAE</th>
<th align="left">RMSE</th>
<th align="left">CSI</th>
<th align="left">POD</th>
<th align="left">FAR</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">1</td>
<td align="left">K&#x2009;&#x003D;&#x2009;1</td>
<td align="left">35.01</td>
<td align="left">43.18</td>
<td align="left">61.42</td>
<td align="left"><bold>83.38</bold></td>
<td align="left">29.97</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;2</td>
<td align="left"><bold>34.56</bold></td>
<td align="left"><bold>42.79</bold></td>
<td align="left">61.71</td>
<td align="left">81.95</td>
<td align="left">28.55</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;3</td>
<td align="left">34.68</td>
<td align="left">42.90</td>
<td align="left">61.76</td>
<td align="left">81.92</td>
<td align="left">28.46</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;4</td>
<td align="left">34.66</td>
<td align="left">42.98</td>
<td align="left">61.71</td>
<td align="left">81.54</td>
<td align="left"><bold>28.26</bold></td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;5</td>
<td align="left">34.57</td>
<td align="left">42.82</td>
<td align="left"><bold>61.85</bold></td>
<td align="left">82.02</td>
<td align="left">28.40</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;6</td>
<td align="left">34.84</td>
<td align="left">43.11</td>
<td align="left">61.71</td>
<td align="left">81.99</td>
<td align="left">28.55</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;7</td>
<td align="left">34.90</td>
<td align="left">43.19</td>
<td align="left">61.67</td>
<td align="left">82.09</td>
<td align="left">28.70</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;8</td>
<td align="left">34.90</td>
<td align="left">43.21</td>
<td align="left">61.62</td>
<td align="left">82.31</td>
<td align="left">28.95</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">K&#x2009;&#x003D;&#x2009;1</td>
<td align="left">25.25</td>
<td align="left">31.96</td>
<td align="left">51.39</td>
<td align="left">63.52</td>
<td align="left">27.04</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;2</td>
<td align="left">25.24</td>
<td align="left">31.95</td>
<td align="left">51.26</td>
<td align="left">63.37</td>
<td align="left">27.03</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;3</td>
<td align="left">25.33</td>
<td align="left">32.04</td>
<td align="left">51.16</td>
<td align="left">63.11</td>
<td align="left">26.97</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;4</td>
<td align="left">25.26</td>
<td align="left">31.97</td>
<td align="left">51.37</td>
<td align="left">63.57</td>
<td align="left">27.16</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;5</td>
<td align="left">25.26</td>
<td align="left">31.96</td>
<td align="left">51.44</td>
<td align="left">63.82</td>
<td align="left">27.34</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;6</td>
<td align="left"><bold>25.15</bold></td>
<td align="left"><bold>31.88</bold></td>
<td align="left"><bold>51.57</bold></td>
<td align="left">62.93</td>
<td align="left"><bold>26.56</bold></td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;7</td>
<td align="left">25.17</td>
<td align="left">31.88</td>
<td align="left">51.20</td>
<td align="left">62.99</td>
<td align="left">26.75</td>
</tr>
<tr>
<td></td>
<td align="left">K&#x2009;&#x003D;&#x2009;8</td>
<td align="left">25.37</td>
<td align="left">32.03</td>
<td align="left">51.52</td>
<td align="left"><bold>64.29</bold></td>
<td align="left">27.77</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s6"><label>6</label><title>Conclusion</title>
<p>This article proposes a new spatiotemporal modeling method GAT-GRU to achieve PM2.5 concentration prediction. GAT-GRU model integrates two deep learning methods: graph attention network and the gated recurrent unit, which can accurately and effectively model the temporal and spatial dependence of air pollution monitoring data in different cities. In addition, we also consider the influence of meteorological knowledge on PM2.5 concentration when building the model. Our model learns the temporal and spatial dependence of different cities and incorporates the meteorological attributes of different cities. The results on real-world datasets prove that the GAT-GRU model has excellent predictive performance. The method we propose can be used to predict urban air pollutants to help solve the problems caused by air pollution. In this paper, different types of meteorological features are used as a whole for the input of prediction model. There is no specific analysis for effect of different types of meteorological features on PM2.5 prediction. In the future, we will use the grpah neural network to study the effects of different types of meteorological features on PM2.5 concentration prediction to help achieve more accurate prediction results.</p>
</sec>
</body>
<back>
<fn-group>
<fn fn-type="other"><p><bold>Funding Statement:</bold> Authors The research project is partially supported by National Natural Science Foundation of China under Grant No. 62072015, U19B2039, U1811463. National Key R&#x0026;D Program of China 2018YFB1600903.</p></fn>
<fn fn-type="conflict"><p><bold>Conflicts of Interest:</bold> The authors declare that they have no conflicts of interest to report regarding the present study.</p></fn>
</fn-group>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. A.</given-names> <surname>Weber</surname></string-name>, <string-name><given-names>T. Z.</given-names> <surname>Insaf</surname></string-name>, <string-name><given-names>E. S.</given-names> <surname>Hall</surname></string-name>, <string-name><given-names>T. O.</given-names> <surname>Talbot</surname></string-name> and <string-name><given-names>A. K.</given-names> <surname>Huff</surname></string-name></person-group>, &#x201C;<article-title>Assessing the impact of fine particulate matter (PM2.5) on respiratorycardiovascular chronic diseases in the New York city metropolitan area using hierarchical Bayesian model estimates</article-title>,&#x201D; <source>Environmental Research</source>, vol. <volume>151</volume>, pp. <fpage>399</fpage>&#x2013;<lpage>409</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>C. A.</given-names> <surname>Pope</surname></string-name> and <string-name><given-names>D. W.</given-names> <surname>Dockery</surname></string-name></person-group>, &#x201C;<article-title>Air pollution and life expectancy in China and beyond</article-title>,&#x201D; in <conf-name>Proceedings of the National Academy of Sciences</conf-name>, vol. <volume>110</volume>, no. <issue>32</issue>, pp. <fpage>12861</fpage>&#x2013;<lpage>12862</lpage>, <year>2013</year>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Younan</surname></string-name>, <string-name><given-names>X. H.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Casanova</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Barnard</surname></string-name>, <string-name><given-names>S. A.</given-names> <surname>Gaussoin</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>PM2.5 associated with gray matter atrophy reflecting increased Alzheimer risk in older women</article-title>,&#x201D; <source>Neurology</source>, vol. <volume>96</volume>, no. <issue>8</issue>, pp. <fpage>e1190</fpage>&#x2013;<lpage>e1201</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Hochreiter</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Schmidhuber</surname></string-name></person-group>, &#x201C;<article-title>Long short-term memory</article-title>,&#x201D; <source>Neural Computation</source>, vol. <volume>9</volume>, no. <issue>8</issue>, pp. <fpage>1735</fpage>&#x2013;<lpage>1780</lpage>, <year>1997</year>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>X. D.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>Q.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>Y. Y.</given-names> <surname>Zou</surname></string-name> and <string-name><given-names>G. Z.</given-names> <surname>Wang</surname></string-name></person-group>, &#x201C;<article-title>A Self-organizing lstm-based approach to PM2.5 forecast</article-title>,&#x201D; in <conf-name>Int. Conf. on Cloud Computing and Security</conf-name>, <publisher-name>Springer, Cham</publisher-name>, pp. <fpage>683</fpage>&#x2013;<lpage>693</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>Y. T.</given-names> <surname>Tsai</surname></string-name>, <string-name><given-names>Y. R.</given-names> <surname>Zeng</surname></string-name> and <string-name><given-names>Y. S.</given-names> <surname>Chang</surname></string-name></person-group>, &#x201C;<article-title>Air pollution forecasting using rnn with lstm</article-title>,&#x201D; in <conf-name>IEEE 16th Intl Conf on Dependable, Autonomic and Secure Computing, 16th Intl Conf on Pervasive Intelligence and Computing, 4th Intl Conf on Big Data Intelligence and Computing and Cyber Science and Technology Congress (DASC/PiCom/DataCom/CyberSciTech)</conf-name>, Athens, Greece, pp. <fpage>1074</fpage>&#x2013;<lpage>1079</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Hua</surname></string-name> and <string-name><given-names>X.</given-names> <surname>Wu</surname></string-name></person-group>, &#x201C;<article-title>A hybrid cnn-lstm model for forecasting particulate matter (PM2.5)</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>8</volume>, pp. <fpage>26933</fpage>&#x2013;<lpage>26940</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C. J.</given-names> <surname>Huang</surname></string-name> and <string-name><given-names>P. H.</given-names> <surname>Kuo</surname></string-name></person-group>, &#x201C;<article-title>A deep cnn-lstm model for particulate matter (PM2.5) forecasting in smart cities</article-title>,&#x201D; <source>Sensors</source>, vol. <volume>7</volume>, pp. <fpage>2220</fpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. Z.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Xie</surname></string-name>, <string-name><given-names>J. C.</given-names> <surname>Ren</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Guo</surname></string-name>, <string-name><given-names>Y. Y.</given-names> <surname>Yang</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>Urban PM2.5 concentration prediction via attention-based cnn&#x2013;lstm</article-title>,&#x201D; <source>Applied Sciences</source>, vol. <volume>10</volume>, no. <issue>6</issue>, pp. <fpage>1953</fpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>P.</given-names> <surname>Veli&#x010D;kovi&#x0107;</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Cucurull</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Casanova</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Romero</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Li&#x00F2;</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>Graph attention networks</article-title>,&#x201D; in <conf-name>Int. Conf. on Learning Representations(ICLR)</conf-name>, Vancouver, Canada, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Dey</surname></string-name> and <string-name><given-names>F. M.</given-names> <surname>Salem</surname></string-name></person-group>, &#x201C;<article-title>Gate-variants of gated recurrent unit (GRU) neural networks</article-title>,&#x201D; in <conf-name>IEEE 60th Int. Midwest Symp. on Circuits and Systems (MWSCAS)</conf-name>, Boston, USA, pp. <fpage>1597</fpage>&#x2013;<lpage>1600</lpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>X. R.</given-names> <surname>Shao</surname></string-name> and <string-name><given-names>C. S.</given-names> <surname>Kim</surname></string-name></person-group>, &#x201C;<article-title>Accurate multi-site daily-ahead multi-step pm2.5 concentrations forecasting using space-shared cnn-lstm</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>70</volume>, no. <issue>3</issue>, pp. <fpage>5143</fpage>&#x2013;<lpage>5160</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Fang</surname></string-name>, <string-name><given-names>F. H.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>V. S.</given-names> <surname>Sheng</surname></string-name> and <string-name><given-names>Y. W.</given-names> <surname>Ding</surname></string-name></person-group>, &#x201C;<article-title>A method for improving cnn-based image recognition using DCGAN</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>57</volume>, no. <issue>1</issue>, pp. <fpage>167</fpage>&#x2013;<lpage>178</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>X. R.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>W. F.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Sun</surname></string-name>, <string-name><given-names>X. M.</given-names> <surname>Sun</surname></string-name> and <string-name><given-names>S. K.</given-names> <surname>Jha</surname></string-name></person-group>, &#x201C;<article-title>A robust 3-D medical watermarking based on wavelet transform for data protection</article-title>,&#x201D; <source>Computer Systems Science &#x0026; Engineering</source>, vol. <volume>41</volume>, no. <issue>3</issue>, pp. <fpage>1043</fpage>&#x2013;<lpage>1056</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>X. R.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Sun</surname></string-name>, <string-name><given-names>X. M.</given-names> <surname>Sun</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Sun</surname></string-name> and <string-name><given-names>S. K.</given-names> <surname>Jha</surname></string-name></person-group>, &#x201C;<article-title>Robust reversible audio watermarking scheme for telemedicine and privacy protection</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>71</volume>, no. <issue>2</issue>, pp. <fpage>3035</fpage>&#x2013;<lpage>3050</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Qin</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Yu</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Zou</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Yong</surname></string-name>, <string-name><given-names>Q.</given-names> <surname>Zhao</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>A novel combined prediction scheme based on cnn and lstm for urban PM2.5 concentration</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>7</volume>, pp. <fpage>20050</fpage>&#x2013;<lpage>20059</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>U.</given-names> <surname>Pak</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Ma</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Ryu</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Ryom</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Juhyok</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>Deep learning-based PM2.5 prediction considering the spatiotemporal correlations: A case study of Beijing, China</article-title>,&#x201D; <source>Science of the Total Environment</source>, vol. <volume>699</volume>, pp. <fpage>133561</fpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Fang</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Pang</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Yi</surname></string-name> and <string-name><given-names>V. S.</given-names> <surname>Sheng</surname></string-name></person-group>, &#x201C;<article-title>AttEF: Convolutional lstm encoder-forecaster with attention module for precipitation nowcasting</article-title>,&#x201D; <source>Intelligent Automation &#x0026; Soft Computing</source>, vol. <volume>30</volume>, no. <issue>2</issue>, pp. <fpage>453</fpage>&#x2013;<lpage>466</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Z.</given-names> <surname>Wu</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Pan</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Long</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Zhang</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>A comprehensive survey on graph neural networks</article-title>,&#x201D; <source>IEEE Transactions on Neural Networks and Learning Systems</source>, vol. <volume>32</volume>, no. <issue>1</issue>, pp. <fpage>4</fpage>&#x2013;<lpage>24</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>Y. R.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>Q. Y.</given-names> <surname>Meng</surname></string-name>, <string-name><given-names>L. W.</given-names> <surname>Meng</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>PM2.5-GNN: A domain knowledge enhanced graph neural network for PM2.5 forecasting</article-title>,&#x201D; in <conf-name>Proc. of the 28th Int. Conf. on Advances in Geographic Information Systems</conf-name>, Seattle, USA, pp. <fpage>163</fpage>&#x2013;<lpage>166</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>X.</given-names> <surname>Gao</surname></string-name> and <string-name><given-names>W. D.</given-names> <surname>Li</surname></string-name></person-group>, &#x201C;<article-title>A Graph-based lstm model for PM2.5 forecasting</article-title>,&#x201D; <source>Atmospheric Pollution Research</source>, vol. <volume>12</volume>, no. <issue>9</issue>, pp. <fpage>101150</fpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>J. H.</given-names> <surname>Xu</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>M. Q.</given-names> <surname>Lv</surname></string-name>, <string-name><given-names>C. Q.</given-names> <surname>Zhan</surname></string-name>, <string-name><given-names>S. J.</given-names> <surname>Chen</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>HighAir: A hierarchical graph neural network-based air quality forecasting method</article-title>,&#x201D; arXiv preprint arXiv: 2101.04264, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y. L.</given-names> <surname>Qi</surname></string-name>, <string-name><given-names>Q.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Karimian</surname></string-name> and <string-name><given-names>D.</given-names> <surname>Liu</surname></string-name></person-group>, &#x201C;<article-title>A hybrid model for spatiotemporal forecasting of PM2.5 based on graph convolutional neural network and long short-term memory</article-title>,&#x201D; <source>Science of the Total Environment</source>, vol. <volume>664</volume>, pp. <fpage>1</fpage>&#x2013;<lpage>10</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Fang</surname></string-name>, <string-name><given-names>F. H.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>V. S.</given-names> <surname>Sheng</surname></string-name> and <string-name><given-names>Y. W.</given-names> <surname>Ding</surname></string-name></person-group>, &#x201C;<article-title>SCENT: A new precipitation nowcasting method based on sparse correspondence and deep neural network</article-title>,&#x201D; <source>Neurocomputing</source>, vol. <volume>448</volume>, pp. <fpage>10</fpage>&#x2013;<lpage>20</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. B.</given-names> <surname>Huang</surname></string-name>, <string-name><given-names>B. X.</given-names> <surname>Li</surname></string-name> and <string-name><given-names>W. Q.</given-names> <surname>Wang</surname></string-name></person-group>, &#x201C;<article-title>A preliminary study on the correlation between PM2. 5 concentration and meteorological conditions in jinan</article-title>,&#x201D; <source>Journal of Marine Meteorology</source>, vol. <volume>40</volume>, pp. <fpage>90</fpage>&#x2013;<lpage>97</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Pan</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Gardiner</surname></string-name>, <string-name><given-names>I.</given-names> <surname>Sargent</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Hare</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>A hybrid mlp-cnn classifier for very fine resolution remotely sensed image classification</article-title>,&#x201D; <source>Isprs Journal of Photogrammetry &#x0026; Remote Sensing</source>, vol. <volume>140</volume>, pp. <fpage>133</fpage>&#x2013;<lpage>144</lpage>, <year>2018</year>.</mixed-citation></ref>
</ref-list>
</back>
</article>