<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">69373</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2025.069373</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Bi-STAT&#x002B;: An Enhanced Bidirectional Spatio-Temporal Adaptive Transformer for Urban Traffic Flow Forecasting</article-title>
<alt-title alt-title-type="left-running-head">Bi-STAT&#x002B;: An Enhanced Bidirectional Spatio-Temporal Adaptive Transformer for Urban Traffic Flow Forecasting</alt-title>
<alt-title alt-title-type="right-running-head">Bi-STAT&#x002B;: An Enhanced Bidirectional Spatio-Temporal Adaptive Transformer for Urban Traffic Flow Forecasting</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Cao</surname><given-names>Yali</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Hu</surname><given-names>Weijian</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-3" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Li</surname><given-names>Lingfang</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><email>lingfangli@imust.edu.cn</email></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Li</surname><given-names>Minchao</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Xu</surname><given-names>Meng</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Han</surname><given-names>Ke</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<aff id="aff-1"><label>1</label><institution>Digital Intelligence Industry Academy, Inner Mongolia University of Science and Technology</institution>, <addr-line>Baotou, 014010</addr-line>, <country>China</country></aff>
<aff id="aff-2"><label>2</label><institution>School of Transportation and Logistics, Southwest Jiaotong University</institution>, <addr-line>Chengdu, 611756</addr-line>, <country>China</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Lingfang Li. Email: <email>lingfangli@imust.edu.cn</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2025</year></pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>09</day><month>12</month><year>2025</year></pub-date>
<volume>86</volume>
<issue>2</issue>
<fpage>1</fpage>
<lpage>23</lpage>
<history>
<date date-type="received">
<day>21</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 The Authors.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_69373.pdf"></self-uri>
<abstract>
<p>Traffic flow prediction constitutes a fundamental component of Intelligent Transportation Systems (ITS), playing a pivotal role in mitigating congestion, enhancing route optimization, and improving the utilization efficiency of roadway infrastructure. However, existing methods struggle in complex traffic scenarios due to static spatio-temporal embedding, restricted multi-scale temporal modeling, and weak representation of local spatial interactions. This study proposes Bi-STAT&#x002B;, an enhanced bidirectional spatio-temporal attention framework to address existing limitations through three principal contributions: (1) an adaptive spatio-temporal embedding module that dynamically adjusts embeddings to capture complex traffic variations; (2) frequency-domain analysis in the temporal dimension for simultaneous high-frequency details and low-frequency trend extraction; and (3) an agent attention mechanism in the spatial dimension that enhances local feature extraction through dynamic weight allocation. Extensive experiments were performed on four distinct datasets, including two publicly benchmark datasets (PEMS04 and PEMS08) and two private datasets collected from Baotou and Chengdu, China. The results demonstrate that Bi-STAT&#x002B; consistently outperforms existing methods in terms of MAE, RMSE, and MAPE, while maintaining strong robustness against missing data and noise. Furthermore, the results highlight that prediction accuracy improves significantly with higher sampling rates, providing crucial insights for optimizing real-world deployment scenarios.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Traffic flow prediction</kwd>
<kwd>spatio-temporal feature modeling</kwd>
<kwd>transformer</kwd>
<kwd>intelligent transportation</kwd>
<kwd>deep learning</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>Inner Mongolia Natural Science Foundation</funding-source>
<award-id>2024QN06017</award-id>
<award-id>2025MS06022</award-id>
</award-group>
<award-group id="awg2">
<funding-source>Basic Scientific Research Business Fee Project for Universities in Inner Mongolia</funding-source>
<award-id>2023XKJX019</award-id>
<award-id>2023XKJX024</award-id>
</award-group>
<award-group id="awg3">
<funding-source>Central Guidance on Local Science and Technology Development</funding-source>
<award-id>2024ZY0084</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Intelligent Transportation Systems (ITS) [<xref ref-type="bibr" rid="ref-1">1</xref>] form the backbone of modern urban traffic management. By integrating the Internet of Things, big data, and artificial intelligence, ITS substantially improves traffic efficiency, safety, and sustainability. Traffic flow prediction is a key enabler for upgrading ITS intelligence. Within the ITS framework, large-scale deployments of traffic sensors&#x2014;such as geomagnetic detectors and high-definition cameras&#x2014;collect real-time, multidimensional parameters including vehicle speed, flow, and density, providing a robust data foundation for prediction. Using these dynamic data, traffic flow prediction systems [<xref ref-type="bibr" rid="ref-2">2</xref>] build high-precision models by mining spatio-temporal evolution patterns and applying advanced algorithms in machine learning and deep learning. This capability supports applications such as traffic signal optimization, congestion warnings, dynamic route planning, and emergency response, ultimately enhancing efficiency, safety, and convenience for travelers.</p>
<p>Traffic flow prediction research faces the fundamental challenge of accurately modeling complex spatio-temporal coupling relationships. Deep learning has emerged as the dominant paradigm in this field owing to its exceptional capability for automatic feature extraction. Current research advances primarily focus on three key methodological directions. Time series modeling methods [<xref ref-type="bibr" rid="ref-3">3</xref>&#x2013;<xref ref-type="bibr" rid="ref-6">6</xref>] conceptualize traffic flow as a dynamic evolutionary process, employing sophisticated time dependence analysis to capture multi-scale patterns including short-term correlations, medium-term periodicity, and long-term trends. Spatial topological modeling methods [<xref ref-type="bibr" rid="ref-7">7</xref>,<xref ref-type="bibr" rid="ref-8">8</xref>] utilize network topology representations to quantify interdependencies among traffic entities through node-edge graph structures and spatial propagation analysis. Most notably, spatio-temporal joint modeling methods [<xref ref-type="bibr" rid="ref-9">9</xref>&#x2013;<xref ref-type="bibr" rid="ref-11">11</xref>] have become the prevailing approach, establishing unified representations that concurrently capture temporal dynamics and spatial correlations, thereby enabling comprehensive modeling of system-wide evolutionary patterns and significantly advancing prediction accuracy.</p>
<p>Despite the advances of spatio-temporal Transformer [<xref ref-type="bibr" rid="ref-12">12</xref>] models in traffic flow prediction, three major limitations remain. First, traffic flow exhibits strong dynamics and uncertainty, especially under emergencies, where abrupt state changes can occur within short periods. Most existing models [<xref ref-type="bibr" rid="ref-13">13</xref>] employ static spatio-temporal encoding during feature embedding, limiting their ability to adaptively capture nonlinear dynamics and sudden changes. This constraint reduces prediction accuracy and robustness in complex scenarios. Second, traffic flow contains prominent diurnal and weekly periodic patterns, as well as sudden fluctuations under noise. While self-attention mechanisms [<xref ref-type="bibr" rid="ref-14">14</xref>] can capture temporal correlations, they often lack explicit multi-scale feature modeling. As a result, attention weights are dispersed, leading to incomplete representations of complex dynamic behaviors. Third, local spatial dynamic dependencies are critical for accurate predictions. Although multi-head attention [<xref ref-type="bibr" rid="ref-15">15</xref>] can model global dependencies, it tends to suffer from weight dispersion when handling long sequences or high-dimensional data, weakening the capture of fine-grained local correlations in road networks and lowering the efficiency of spatial feature utilization.</p>
<p>To address these limitations, we propose Bi-STAT&#x002B;, an enhanced bidirectional spatio-temporal attention model that augments embedding representation, temporal modeling, and spatial modeling in the Bi-STAT framework. In the embedding stage, we introduce an adaptive reconciliation mechanism to dynamically adjust spatial and temporal embeddings, enabling the model to capture nonlinear features and abrupt traffic patterns. For temporal modeling, we incorporate a frequency-domain analysis to transform time series into the spectral representation, facilitating joint extraction of high-frequency details (e.g., short-term fluctuations) and low-frequency trends (e.g., daily/weekly cycles). For spatial modeling, we design an Agent Attention mechanism that strengthens the local correlations extraction&#x2014;such as interactions between adjacent sensors&#x2014;through agent vectors and dynamic weight allocation.</p>
<p>The main contributions are as follows:
<list list-type="order">
<list-item>
<p>We propose Bi-STAT&#x002B;, a model for urban traffic flow forecasting that enhances spatio-temporal representation and improves prediction accuracy.</p>
</list-item>
<list-item>
<p>We develop a spatio-temporal adaptive embedding module that dynamically adjusts spatial and temporal embeddings to capture nonlinear features and abrupt changes, thereby enhancing robustness under complex traffic conditions.</p>
</list-item>
<list-item>
<p>We upgrade the spatio-temporal adaptive Transformer by integrating a Frequency Domain Enhanced Temporal Adaptive Transformer (FDETAT) and a Spatial Adaptive Agent Transformer (SAAT), improving spatial and temporal dependency modeling.</p>
</list-item>
<list-item>
<p>We validate Bi-STAT&#x002B; on public datasets PEMS04, PEMS08, and private datasets from Baotou and Chengdu, China. The results show that our model achieves superior performance in MAE, RMSE, and MAPE, exhibits strong robustness to missing data and noise, and further reveal that higher sensor sampling frequency improves prediction accuracy, providing practical guidance for real-world deployment.</p>
</list-item>
</list></p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Work</title>
<sec id="s2_1">
<label>2.1</label>
<title>Time Series Modeling Methods</title>
<p>Time series modeling method is a core component in traffic flow prediction, aiming to capture the dynamic evolution of traffic patterns. Early RNN-based approaches successfully modeled sequential features, but struggled with long-term dependencies (e.g., shifts from morning peak to midday off-peak periods) due to vanishing gradients. LSTMs addressed this limitation through gate mechanisms: the forget gate filters redundant information during low-flow periods, while the input and output gates preserve peak-flow features. For instance, the D-STN model [<xref ref-type="bibr" rid="ref-3">3</xref>] integrates convolutional operations with gating mechanisms, improving accuracy in modeling 12-h traffic flow trends on urban main roads. Faraz Malik Awan et al. [<xref ref-type="bibr" rid="ref-16">16</xref>] applied LSTM to multi-source data, integrating traffic flow and noise concentration, thereby reducing short-term prediction errors under rainy conditions. GRUs simplify the architecture while preserving essential functionality, making them more effective for short-term, high-frequency fluctuation scenarios. DeepTP [<xref ref-type="bibr" rid="ref-5">5</xref>] employs GRU to improve short-term prediction for 5-min interval data, while STGNN-FAM [<xref ref-type="bibr" rid="ref-6">6</xref>] uses bidirectional GRUs to capture morning and evening rush-hour patterns, enhancing accuracy during transition periods.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Spatial Topological Modeling Methods</title>
<p>Beyond temporal dependency, complex spatial correlations within transportation networks are equally critical for traffic flow prediction. Early approaches introduced CNNs, which were originally designed for image processing to capture spatial patterns. However, the inherent non-Euclidean structure of transportation networks fundamentally limits grid-based representations, inducing topological distortion and connectivity loss. Consequently, graph-based modeling methods have become mainstream. GCNs capture non-Euclidean spatial dependencies by defining convolution operations directly on graph structures. DCRNN [<xref ref-type="bibr" rid="ref-17">17</xref>] employs diffusion convolution to model traffic flow propagation in directed graphs, enhancing spatial correlation representation. However, its reliance on fixed adjacency matrices fails to adapt to structural changes in dynamic traffic environments. To address this, T-GCN [<xref ref-type="bibr" rid="ref-18">18</xref>] integrates graph convolution for spatial feature extraction, whereas AGCRN [<xref ref-type="bibr" rid="ref-7">7</xref>] leverages adaptive graph learning to dynamically generate adjacency matrices. DSTGCN [<xref ref-type="bibr" rid="ref-19">19</xref>] constructs dynamic spatio-temporal graphs to characterize road network interactions, substantially improving accuracy. Despite these advances, modeling large-scale dynamic graphs remains computationally intensive.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Spatio-Temporal Joint Modeling Methods</title>
<p>Modeling exclusively either the temporal or spatial dimension cannot fully capture traffic flow complexity. Consequently, spatio-temporal fusion methods have emerged as a prominent research focus. Early approaches cascaded temporal and spatial modules sequentially, whereas advanced methods employ spatio-temporal graph convolutional networks to jointly model spatial adjacency and temporal dynamics. ST-CGCN [<xref ref-type="bibr" rid="ref-10">10</xref>] applies dynamic graph convolution for spatial feature extraction and LSTM for temporal dependencies, while leveraging complex graph structures to represent nonlinear interactions. STFGCN [<xref ref-type="bibr" rid="ref-11">11</xref>] introduces hierarchical graph convolution with adaptive fusion to model multi-level associations efficiently. The emergence of Transformers has significantly advanced traffic prediction. Transformer use multi-head self-attention to model global dependencies, enabling parallel long-sequence processing while overcoming the vanishing gradient and sequential computation limitations inherent to RNN. Unlike recursive architectures, Transformer excel in modeling non-stationary patterns and complex spatio-temporal dependencies. For instance, Bi-STAT [<xref ref-type="bibr" rid="ref-15">15</xref>] separates spatial and temporal Transformer modules, Traffic Transformer [<xref ref-type="bibr" rid="ref-13">13</xref>] integrates GCNs for better spatial perception, and GMAN [<xref ref-type="bibr" rid="ref-14">14</xref>] adopts graph-based multi-head attention to enhance dynamic pattern representation.</p>
<p>Although Transformer effectively capture global spatio-temporal dependencies, they model frequency-domain features inadequately. Multi-frequency information is essential: low-frequency components reflect long-term trends, whereas high-frequency components capture short-term fluctuations and anomalies. Traditional Transformer emphasize low-frequency trends but often diminish high-frequency signals [<xref ref-type="bibr" rid="ref-20">20</xref>]. In contrast, convolutional architectures such as TCNs better preserve high-frequency details [<xref ref-type="bibr" rid="ref-21">21</xref>]. To bridge this gap, Feng et al. [<xref ref-type="bibr" rid="ref-22">22</xref>] proposed a two-layer routing attention mechanism combining patch merging for multi-scale low-frequency extraction with MLPs for short-term high-frequency variation, achieving preliminary frequency-domain fusion. Nonetheless, fully integrating multi-frequency features while retaining spatio-temporal dependency modeling remains an urgent challenge.</p>
<p>Traffic flow prediction research has progressed from single-dimensional to multi-dimensional, multi-module collaborative modeling, employing methods such as RNNs, GCNs, STGCNs, and Transformers. These approaches have advanced spatio-temporal feature extraction, global dependency modeling, and dynamic structure learning. Nevertheless, persistent challenges impede the effective integration of multi-frequency components and the insufficient representation of dynamic relationships. These gaps necessitate an enhanced framework and form the theoretical foundation for the Bi-STAT&#x002B; model proposed in this study.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Methodology</title>
<sec id="s3_1">
<label>3.1</label>
<title>Framework Overview</title>
<p>The Bi-STAT&#x002B; model builds upon the Encoder&#x2013;Decoder architecture of Bi-STAT [<xref ref-type="bibr" rid="ref-15">15</xref>], optimizing critical components of spatio-temporal feature modeling in traffic flow prediction to address the limitations of existing methods.</p>
<p>As illustrated in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>, Bi-STAT&#x002B; consists of a Spatio-Temporal Adaptive Embedding (STAE) module, an encoder, a cross-attention module, and a decoder. STAE module enhances adaptability to spatio-temporal features and improves the efficiency of capturing complex spatio-temporal dependencies. Both encoder and decoder upgrade the original temporal and spatial Transformer structures to better address spatial heterogeneity and temporal multiscale characteristics in traffic flow prediction. The encoder contains multiple Spatio-Temporal Adaptive Transformer units, each of which integrates a Spatial Adaptive Agent Transformer (SAAT), a Frequency Domain Enhanced Temporal Adaptive Transformer (FDETAT), and an entanglement module for modeling spatio-temporal sequence interactions. The decoder adopts a dual-branch architecture that includes a prediction branch and a recall branch. The prediction branch performs similar functions to the encoder. Notably, the recall branch regularizes the model by learning historical traffic representations, omitting the DHM module to focus more on future traffic flow prediction.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Structure of Bi-STAT&#x002B;</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_69373-fig-1.tif"/>
</fig>
<p>Bi-STAT&#x002B; accepts road network spatial topology and traffic flow data as inputs, denoted by <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mi>X</mml:mi></mml:math></inline-formula>. First, the STAE module learns efficient feature representations from the input, producing representative spatio-temporal features <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mi>S</mml:mi><mml:mi>T</mml:mi><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. The encoder then jointly models spatial topology and temporal sequences, capturing complex dependencies in traffic flows to generate <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. Based on this, the cross-attention module establishes an efficient information exchange between the encoder and decoder, comprising past&#x2013;present and present&#x2013;future cross-attention branches. The past&#x2013;present branch outputs <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, enabling the decoder to fully exploit historical information and mitigate error accumulation. The present&#x2013;future branch outputs <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, assisting in accurate future traffic prediction. Finally, the model is trained using a dual-branch structure comprising a prediction decoder and a recall decoder. The prediction decoder focuses on future traffic flow, outputting <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. The recall decoder reconstructs historical information to reduce overfitting, outputting <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. This design enhances robustness and predictive accuracy, ultimately enabling high-precision traffic flow forecasts.</p>
<p>Overall, compared with the original Bi-STAT, Bi-STAT&#x002B; replaces the original Temporal Adaptive Transformer with FDETAT, enhancing the modeling of multiscale temporal features. Instead of the original Spatial Adaptive Transformer, it employs SAAT to dynamically capture local and global interactions among road network nodes, improving the efficiency of spatial dependency modeling.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Spatio-Temporal Adaptive Embedding</title>
<p>The STAE module takes temporal data and road network structural data as input, and generates a unified feature representation for the model. As shown in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>, the module is composed of spatial embedding, adaptive reconciliation, and temporal embedding. The implementation details are provided in Algorithm 1.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Spatio-temporal adaptive embedding module</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_69373-fig-2.tif"/>
</fig>
<fig id="fig-10">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_69373-fig-10.tif"/>
</fig>
<p><bold>The spatial embedding branch</bold> captures spatial characteristics of the road network to model complex spatial dependencies among traffic sensors. A traffic network graph is constructed from actual road network data, where nodes denote traffic sensors and edges indicate road connectivity. Pairwise distances between all traffic sensors are then computed to better reflect actual traffic flow propagation paths. These distances are normalized using a Gaussian kernel function [<xref ref-type="bibr" rid="ref-23">23</xref>] to generate an adjacency matrix that captures spatial correlations between sensors. The node2vec method [<xref ref-type="bibr" rid="ref-24">24</xref>] maps graph nodes (traffic sensors) into low-dimensional feature representations for spatial embedding. These embeddings are then processed through two fully connected layers to align them with the target output dimension, yielding the spatial embedding <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>.</p>
<p><bold>The adaptive reconciliation branch</bold> does not process input data directly but provides a dynamically adjustable embedding mechanism to enhance the extraction of complex spatio-temporal features. It optimizes embeddings based on data variations by training a reconciliation matrix, improving adaptation to spatio-temporal dependencies in diverse scenarios. Specifically, a learnable parameter tensor of shape (time length, number of nodes, adaptive embedding dimension) is created as the reconciliation matrix and initialized using the Xavier method [<xref ref-type="bibr" rid="ref-25">25</xref>]. The reconciliation matrix is then transformed through two fully connected layers to adjust its dimension and map it to the target output dimension. Finally, the Mean module averages the feature tensors&#x2014;output from the fully connected layers&#x2014;over spatial dimensions (different locations) and temporal dimensions (historical, current, and future steps), yielding the adaptive reconciliation representation <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> with consistent dimensions.</p>
<p><bold>The temporal embedding branch</bold> captures periodic patterns in traffic data to model recurring traffic flow behaviors. For each time step, two period embedding matrices are generated: day-of-week and time-of-day. These matrices represent weekly and daily periodic features, respectively. They are concatenated to form a comprehensive periodic feature matrix, integrating information across multiple time scales. The concatenated matrix is then fed into two fully connected layers. After a nonlinear transformation, the features are mapped to the target output dimension, producing the periodic embedding <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>.</p>
<p>The spatial embedding <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, adaptive reconciliation <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, and temporal embedding <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are summed to produce the fused spatio-temporal embedding <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:mi>S</mml:mi><mml:mi>T</mml:mi><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. This fusion enables <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> to dynamically adjust <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, learning optimal feature representations from input context during training. This mechanism strengthens spatio-temporal embedding capability, thereby improving predictive performance.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Frequency Domain Enhanced Temporal Adaptive Transformer</title>
<p>Traffic flow data typically exhibit complex long-term trends alongside short-term fluctuations. However, the original Temporal Adaptive Transformer primarily models overall temporal dynamics, limiting its ability to capture features across multiple time scales. Moreover, traffic data often contain high-frequency noise (e.g., unexpected events, missing data), which distorts short-term trends and degrades overall prediction accuracy, thereby reducing model robustness and stability. To address these limitations, the FDETAT extends the original structure by integrating the Frequency-Enhanced Channel Attention Mechanism (FECAM) [<xref ref-type="bibr" rid="ref-26">26</xref>]. This integration strengthens the model&#x2019;s ability to capture both long- and short-term traffic flow features via frequency-domain enhancement.</p>
<p>As shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>, the FDETAT comprises five key components: Multi-Head Attention, FECAM, Add &#x0026; Norm, Transition Function, and DHM. The architecture takes the temporal feature <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> as input and learns global dependencies through Multi-Head Attention, capturing correlations between different time steps. Subsequently, FECAM extracts both long- and short-term temporal features, enhancing the model&#x2019;s perception of diverse temporal patterns. A residual connection followed by normalization transforms the temporal features into optimized representations, denoted as <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. The Transition Function module then adjusts the feature space to maximize information flow, enabling efficient transformations between temporal features. Finally, the DHM component selectively halts or continues information delivery, prioritizing task-relevant data. This module improves information-processing efficiency and allows dynamic adjustment of computational steps based on task complexity. Ultimately, the optimized temporal features <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are used in subsequent decoding stages to enhance task-specific performance.</p>
<p>As the core of FDETAT, FECAM processes traffic flow data in the frequency domain using the Discrete Cosine Transform (DCT) [<xref ref-type="bibr" rid="ref-27">27</xref>]. Through this transform, FECAM models temporal characteristics at multiple frequency scales, enhancing the model&#x2019;s ability to capture complex spatio-temporal patterns. Low-frequency components reveal long-term trends and periodic patterns (e.g., day&#x2013;night cycles, peak hours), whereas high-frequency components capture short-term fluctuations and localized changes (e.g., traffic accidents, missing data from sensor failures). FECAM compensates for the limitations of traditional multi-attention mechanisms in frequency-domain information extraction, thereby improving adaptability and robustness in complex traffic scenarios.</p>
<p>The overall architecture of FECAM is illustrated in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>. The input to FECAM is the traffic flow feature <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:mi>T</mml:mi></mml:math></inline-formula>, comprising time series data for <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mi>N</mml:mi></mml:math></inline-formula> nodes, each of length <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mi>L</mml:mi></mml:math></inline-formula>, representing traffic flow observations at discrete time points. Specifically, <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:mi>T</mml:mi></mml:math></inline-formula> is split along the node dimension into <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:mi>N</mml:mi></mml:math></inline-formula> subsequences <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:mrow><mml:mo>{</mml:mo><mml:msub><mml:mi>v</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:msub><mml:mi>v</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:msub><mml:mi>v</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, where each <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:msub><mml:mi>v</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the time series of a single traffic node. Each <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:msub><mml:mi>v</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is transformed using the DCT to obtain its frequency representation <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:mi>F</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:msub><mml:mi>q</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. The DCT maps time-domain data to the frequency domain, enabling the model to capture cyclical patterns&#x2014;information crucial for modeling periodic trends in traffic flow. The frequency representations <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:mrow><mml:mo>{</mml:mo><mml:mi>F</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:msub><mml:mi>q</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>F</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:msub><mml:mi>q</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>F</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:msub><mml:mi>q</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula> from all nodes are stacked to form the complete frequency-domain tensor Freq, preserving frequency information across nodes. This tensor Freq passes through a fully connected layer to learn task-specific frequency-domain features, followed by normalization to produce an importance weight matrix that quantifies the contribution of each frequency component to prediction. The computed importance weights are applied element-wise to <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:mi>T</mml:mi></mml:math></inline-formula>, amplifying beneficial frequency components while suppressing irrelevant or noisy ones. The computational procedure of FECAM is formally defined in <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref>:
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msup><mml:mi>T</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mi>T</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>N</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>F</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>S</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>k</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>D</mml:mi><mml:mi>C</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>S</mml:mi><mml:mi>p</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>T</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>FECAM module</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_69373-fig-3.tif"/>
</fig>
<p>By incorporating attention in the frequency domain, FECAM focuses on critical frequency components in time-series data, thereby improving traffic flow prediction and enhancing the capture of both periodic and trending patterns.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Spatial Adaptive Agent Transformer</title>
<p>Spatial relationships in traffic flow data typically show strong localization, where traffic in a given road segment is mainly influenced by its neighboring segments, and correlations diminish as spatial distance increases. In traditional spatial adaptive transformers, the multi-head attention mechanism assigns weights to all nodes to capture global relationships. However, this global allocation can cause weight dispersion, reducing the ability to effectively capture critical local interactions.</p>
<p>As illustrated in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>, the SAAT model replaces conventional multi-head attention with the Agent Attention mechanism [<xref ref-type="bibr" rid="ref-28">28</xref>]. The module receives the spatial feature matrix <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mi>p</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> as input, applies Agent Attention to adaptively learn the importance of different regions, and dynamically reallocates computational resources, thereby improving its ability to capture both local and global dependencies. After residual concatenation and normalization, the spatial features are transformed into optimized representations <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. A transition module then adjusts the feature representation space to optimize information transfer and ensure effective mapping between spatial dimensions. Finally, the DHM component selectively halts or continues information transfer, enabling the model to dynamically focus on key features relevant to the current task. The resulting optimized spatial feature <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> serves as the final representation for subsequent decoding stages, enhancing performance, particularly in complex spatio-temporal dependency scenarios.</p>
<p>In Agent attention, the traditional attention triplet <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:mo stretchy="false">(</mml:mo><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>V</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is extended to a quadruplet <inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:mo stretchy="false">(</mml:mo><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mspace width="thinmathspace" /><mml:mi>V</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, introducing an additional agent vector <inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mi>A</mml:mi></mml:math></inline-formula> for aggregating information from <inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:mi>K</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:mi>V</mml:mi></mml:math></inline-formula>, and then transferring to <inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:mi>Q</mml:mi></mml:math></inline-formula> to model global information. The overall structure of Agent Attention comprises two standard Softmax Attention operations and is mathematically equivalent to a generalized linear attention mechanism. In this way, Agent attention seamlessly integrates high-performance Softmax Attention with efficient linear attention, maintaining sensitivity to local influences while preserving a global understanding of the entire transportation network.</p>
<p><xref ref-type="fig" rid="fig-4">Fig. 4</xref> depicts the detailed workflow of the Agent attention module. The process begins with a linear transformation of the input traffic network adjacency matrix <inline-formula id="ieqn-50"><mml:math id="mml-ieqn-50"><mml:mi>S</mml:mi></mml:math></inline-formula>, generating the query (<inline-formula id="ieqn-51"><mml:math id="mml-ieqn-51"><mml:mi>Q</mml:mi></mml:math></inline-formula>), key (<inline-formula id="ieqn-52"><mml:math id="mml-ieqn-52"><mml:mi>K</mml:mi></mml:math></inline-formula>), and value (<inline-formula id="ieqn-53"><mml:math id="mml-ieqn-53"><mml:mi>V</mml:mi></mml:math></inline-formula>) matrices for subsequent attention computation. The query matrix <inline-formula id="ieqn-54"><mml:math id="mml-ieqn-54"><mml:mi>Q</mml:mi></mml:math></inline-formula> then undergoes pooling to extract Agent Tokens (<inline-formula id="ieqn-55"><mml:math id="mml-ieqn-55"><mml:mi>A</mml:mi></mml:math></inline-formula>), which aggregate features from multiple sensor nodes. In the first stage, Agent Features are computed as shown in <xref ref-type="disp-formula" rid="eqn-2">Eq. (2)</xref>. The Agent Tokens <inline-formula id="ieqn-56"><mml:math id="mml-ieqn-56"><mml:mi>A</mml:mi></mml:math></inline-formula> serve as new queries in a Softmax Attention operation with the key matrix <inline-formula id="ieqn-57"><mml:math id="mml-ieqn-57"><mml:mi>K</mml:mi></mml:math></inline-formula>, producing Agent Features <inline-formula id="ieqn-58"><mml:math id="mml-ieqn-58"><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> that enhance the model&#x2019;s perception of critical traffic patterns.
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mspace width="negativethinmathspace" /><mml:mi>f</mml:mi><mml:mspace width="negativethinmathspace" /><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>A</mml:mi><mml:msup><mml:mi>K</mml:mi><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:msqrt><mml:mi>d</mml:mi></mml:msqrt></mml:mfrac><mml:mo>+</mml:mo><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:mi>V</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Agent attention module</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_69373-fig-4.tif"/>
</fig>
<p>Next, perform a Softmax Attention operation taking <inline-formula id="ieqn-59"><mml:math id="mml-ieqn-59"><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> as value and <inline-formula id="ieqn-60"><mml:math id="mml-ieqn-60"><mml:mi>A</mml:mi></mml:math></inline-formula> as key, and keeping the original <inline-formula id="ieqn-61"><mml:math id="mml-ieqn-61"><mml:mi>Q</mml:mi></mml:math></inline-formula> as query, which strengthens global feature interactions and yields the intermediate feature <inline-formula id="ieqn-62"><mml:math id="mml-ieqn-62"><mml:mi>O</mml:mi></mml:math></inline-formula>, as <xref ref-type="disp-formula" rid="eqn-3">Eq. (3)</xref> shows.
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>O</mml:mi><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mspace width="negativethinmathspace" /><mml:mi>f</mml:mi><mml:mspace width="negativethinmathspace" /><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>Q</mml:mi><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mi>A</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:msqrt><mml:mi>d</mml:mi></mml:msqrt></mml:mfrac><mml:mo>+</mml:mo><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>The final output <inline-formula id="ieqn-63"><mml:math id="mml-ieqn-63"><mml:mi>Y</mml:mi></mml:math></inline-formula> directly fused the intermediate feature <inline-formula id="ieqn-64"><mml:math id="mml-ieqn-64"><mml:mi>O</mml:mi></mml:math></inline-formula> and local feature which is processed by Depthwise Separable Convolution (DWC), as shown in <xref ref-type="disp-formula" rid="eqn-4">Eq. (4)</xref>.
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>Y</mml:mi><mml:mo>=</mml:mo><mml:mi>O</mml:mi><mml:mo>+</mml:mo><mml:mi>D</mml:mi><mml:mi>W</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>This workflow combines a two-stage Softmax Attention mechanism with a local fusion stage via deep convolution, enabling the model to capture both global context and local traffic dynamics.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experiments</title>
<sec id="s4_1">
<label>4.1</label>
<title>Datasets and Preprocessing</title>
<p>This study employed four real-world traffic flow datasets for model validation, comprising two public and two private datasets. The public datasets are the widely used traffic flow prediction benchmarks PeMS04 and PeMS08 [<xref ref-type="bibr" rid="ref-29">29</xref>]. The private datasets, obtained from traffic surveillance systems in Chengdu and Baotou, China, encompass diverse traffic scenarios and complex environmental conditions, thereby enhancing the comprehensiveness and generalizability of model validation. Detailed dataset descriptions are provided in <xref ref-type="table" rid="table-1">Table 1</xref>, while <xref ref-type="fig" rid="fig-5">Fig. 5a</xref>,<xref ref-type="fig" rid="fig-5">b</xref> illustrates the spatial distribution of sensors in the two private datasets.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Dataset information</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Dataset</th>
<th>Sensors</th>
<th>Time range</th>
<th>Time steps</th>
<th>Intervals</th>
<th>Dataset partitioning</th>
</tr>
</thead>
<tbody>
<tr>
<td>PEMS04</td>
<td>307</td>
<td>01 January 2018&#x2013;28 February 2018</td>
<td>16,992</td>
<td>5 min</td>
<td>11,894:1700:3398</td>
</tr>
<tr>
<td>PEMS08</td>
<td>170</td>
<td>01 July 2016&#x2013;31 August 2016</td>
<td>17,856</td>
<td>5 min</td>
<td>12,499:1786:3571</td>
</tr>
<tr>
<td>Baotou</td>
<td>355</td>
<td>07 June 2024&#x2013;07 July 2024</td>
<td>744</td>
<td>1 h</td>
<td>521:74:149</td>
</tr>
<tr>
<td>Chengdu</td>
<td>350</td>
<td>01 January 2022&#x2013;24 February 2022</td>
<td>1320</td>
<td>1 h</td>
<td>924:132:264</td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>(<bold>a</bold>) Distribution of nodes in Baotou; (<bold>b</bold>) Distribution of nodes in Chengdu</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_69373-fig-5.tif"/>
</fig>
<p>During data preprocessing, a comprehensive data cleaning and processing scheme was developed. First, based on integrity analysis, the original monitoring data were screened. Data from monitoring points with high missing rates, along with their corresponding periods, were removed to ensure reliability. Second, for the filtered dataset, missing values were filled using linear interpolation. This method preserves temporal continuity and ensures spatial consistency. To ensure experimental rigor, a stratified sampling strategy was applied, dividing all datasets into training, validation, and test sets in a 7:1:2 ratio based on temporal order (see <xref ref-type="table" rid="table-1">Table 1</xref> for details). The division strictly followed temporal continuity, ensuring no overlap in time between subsets. This approach effectively prevented data leakage and provided a robust foundation for subsequent model training and evaluation.</p>

</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Baselines</title>
<p>To thoroughly assess the predictive performance of the Bi-STAT&#x002B; model, 14 representative benchmark models were selected for comparative experiments. These comprise one traditional statistical model, HA (1997) [<xref ref-type="bibr" rid="ref-30">30</xref>]; five deep learning models based on Graph Neural Networks (GNNs)&#x2014;ASTGCN (2019) [<xref ref-type="bibr" rid="ref-8">8</xref>], AGCRN (2020) [<xref ref-type="bibr" rid="ref-7">7</xref>], STFGNN (2021) [<xref ref-type="bibr" rid="ref-31">31</xref>], DSTAGNN (2022) [<xref ref-type="bibr" rid="ref-32">32</xref>], and ST-CGCN (2023) [<xref ref-type="bibr" rid="ref-10">10</xref>]; and eight deep learning models based on the Transformer architecture&#x2014;GMAN (2020) [<xref ref-type="bibr" rid="ref-14">14</xref>], ASTGNN (2021) [<xref ref-type="bibr" rid="ref-33">33</xref>], Traffic Transformer (2022) [<xref ref-type="bibr" rid="ref-13">13</xref>], Bi-STAT (2022) [<xref ref-type="bibr" rid="ref-15">15</xref>], MFE-STL (2024) [<xref ref-type="bibr" rid="ref-34">34</xref>], STFGCN (2024) [<xref ref-type="bibr" rid="ref-11">11</xref>], GAMAN (2025) [<xref ref-type="bibr" rid="ref-35">35</xref>], and FDGT (2025) [<xref ref-type="bibr" rid="ref-36">36</xref>].</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Experimental Setups</title>
<p>The experimental platform runs on Ubuntu 20.04 LTS, with model implementation based on the PyTorch 2.1.1 deep learning framework. The hardware configuration comprises an Intel&#x00AE; Xeon&#x00AE; E5-2667 v3 processor (3.20 GHz), 24 GB RAM, and an NVIDIA GeForce RTX 3090 GPU (CUDA 12.2).</p>
<p>Following experimental validation and parameter tuning, the model training parameters were configured as follows. The batch size was set to 4 for the PeMS04 and PeMS08 datasets, and to 2 for the larger Baotou and Chengdu datasets. The encoder&#x2013;decoder depth was set to two layers for all datasets, except for PeMS04, where a single layer was used due to its data characteristics. The Adam optimizer was employed with an initial learning rate of 0.001, combined with a ReduceLROnPlateau scheduler for dynamic adjustment. Training was conducted for 100 epochs, with early stopping triggered if validation performance failed to improve for 10 consecutive epochs. The DHM penalty term weight and recall decoder weight were both set to 0.001, and the reconciliation matrix dimension was fixed at 80.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Evaluation Metrics</title>
<p>To assess model prediction performance, three widely used evaluation metrics are employed: mean absolute error (MAE), root mean square error (RMSE), and mean absolute percentage error (MAPE). All three metrics quantify the deviation between predicted and actual values, with smaller values indicating higher prediction accuracy and larger values indicating greater prediction error. The formulas for each metric are provided in <xref ref-type="disp-formula" rid="eqn-5">Eqs. (5)</xref>&#x2013;<xref ref-type="disp-formula" rid="eqn-7">(7)</xref>.
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac><mml:msqrt><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:msqrt></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>P</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mn>100</mml:mn><mml:mi mathvariant="normal">&#x0025;</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Comparison with the SOTA Methods</title>
<sec id="s4_5_1">
<label>4.5.1</label>
<title>Comparative Experiments on Public Datasets</title>
<p>This section evaluates the predictive performance of the proposed Bi-STAT&#x002B; model against multiple benchmark models on the PEMS04 and PEMS08 datasets. In all experiments, the historical step size (H), current step size (P), and forecast step size (F) were set to 12. For datasets in the PEMS series, with a time granularity of 5 min, the model uses the most recent 1 h of historical data to forecast traffic flow for the next hour. The results are summarized in <xref ref-type="table" rid="table-2">Table 2</xref>, where bold values indicate the best performance and underlined values denote the second best. All results are reported as the mean values over 12-step forecasts.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Comparison of the performance of different models in predicting the next hour on the PEMS04 and PEMS08 datasets</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Dataset</th>
<th colspan="3">PEMS04</th>
<th colspan="3">PEMS08</th>
</tr>
<tr>
<th>Metric</th>
<th>MAE&#x2193;</th>
<th>RMSE&#x2193;</th>
<th>MAPE (%)&#x2193;</th>
<th>MAE&#x2193;</th>
<th>RMSE&#x2193;</th>
<th>MAPE (%)&#x2193;</th>
</tr>
</thead>
<tbody>
<tr>
<td>HA</td>
<td>31.06</td>
<td>46.52</td>
<td>23.04</td>
<td>25.63</td>
<td>38.42</td>
<td>16.19</td>
</tr>
<tr>
<td>ASTGCN</td>
<td>22.42</td>
<td>35.23</td>
<td>15.00</td>
<td>18.89</td>
<td>29.11</td>
<td>11.00</td>
</tr>
<tr>
<td>AGCRN</td>
<td>19.85</td>
<td>32.63</td>
<td>13.12</td>
<td>16.33</td>
<td>25.89</td>
<td>10.58</td>
</tr>
<tr>
<td>STFGNN</td>
<td>19.83</td>
<td>31.88</td>
<td>13.02</td>
<td>16.64</td>
<td>26.22</td>
<td>10.60</td>
</tr>
<tr>
<td>DSTAGNN</td>
<td>19.30</td>
<td>31.46</td>
<td>12.70</td>
<td>15.67</td>
<td>24.77</td>
<td>9.94</td>
</tr>
<tr>
<td>ST-CGCN</td>
<td>20.79</td>
<td>33.62</td>
<td>13.71</td>
<td>17.84</td>
<td>26.43</td>
<td>10.63</td>
</tr>
<tr>
<td>GMAN</td>
<td>19.36</td>
<td>31.06</td>
<td>13.55</td>
<td>14.51</td>
<td>23.68</td>
<td>9.45</td>
</tr>
<tr>
<td>ASTGNN</td>
<td><underline>18.65</underline><sup>1</sup></td>
<td>30.91</td>
<td>12.40</td>
<td>15.16</td>
<td>24.71</td>
<td>9.82</td>
</tr>
<tr>
<td>Traffic transformer</td>
<td>19.16</td>
<td>30.57</td>
<td>13.70</td>
<td>15.37</td>
<td>24.21</td>
<td>10.09</td>
</tr>
<tr>
<td>Bi-STAT</td>
<td>18.81</td>
<td><underline>30.38</underline><sup>1</sup></td>
<td>12.72</td>
<td><underline>14.13</underline><sup>1</sup></td>
<td><underline>23.34</underline><sup>1</sup></td>
<td><underline>9.13</underline><sup>1</sup></td>
</tr>
<tr>
<td>MFE-STL</td>
<td>19.22</td>
<td>31.17</td>
<td>12.61</td>
<td>15.48</td>
<td>24.51</td>
<td>9.92</td>
</tr>
<tr>
<td>STFGCN</td>
<td>18.95</td>
<td>30.90</td>
<td><underline>12.36</underline><sup>1</sup></td>
<td>15.23</td>
<td>24.35</td>
<td>9.83</td>
</tr>
<tr>
<td>GAMAN</td>
<td>18.97</td>
<td>30.64</td>
<td>12.71</td>
<td>14.69</td>
<td>23.98</td>
<td>10.03</td>
</tr>
<tr>
<td>FDGT</td>
<td>19.01</td>
<td>31.15</td>
<td>12.75</td>
<td>14.23</td>
<td>23.55</td>
<td>9.56</td>
</tr>
<tr>
<td>Bi-STAT&#x002B; (ours)</td>
<td><bold>18.25</bold><sup>&#x022C6;</sup></td>
<td><bold>29.96</bold><sup>&#x022C6;</sup></td>
<td><bold>12.27</bold><sup>&#x022C6;</sup></td>
<td><bold>13.39</bold><sup>&#x022C6;</sup></td>
<td><bold>22.74</bold><sup>&#x022C6;</sup></td>
<td><bold>8.82</bold><sup>&#x022C6;</sup></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-2fn1" fn-type="other"><p>Note: <sup>&#x022C6;</sup>Bold font indicates the best result; <sup>1</sup>Underlined font indicates the second-best result.</p></fn></table-wrap-foot>
</table-wrap>
<p>On both PEMS04 and PEMS08, model performance varies markedly. Bi-STAT&#x002B; achieves the highest accuracy, whereas the traditional HA model performs the worst. For PEMS04, Bi-STAT&#x002B; reduces MAE by about 4% compared with the latest Transformer-based model, FDGT. For PEMS08, Bi-STAT&#x002B; achieves reductions of approximately 5.2%, 2.6%, and 3.4% in the respective metrics compared with the second-best model, Bi-STAT.</p>
<p>The observed performance differences stem from each model&#x2019;s ability to capture spatio-temporal traffic dynamics. HA relies solely on historical averages and ignores dynamic patterns, resulting in the poorest performance. Graph neural network-based models (e.g., ASTGNN) can capture spatial correlations but struggle with temporal precision and depend partly on fixed network topologies, limiting adaptability to dynamic changes. Transformer-based models (e.g., FDGT) excel in modeling long-range dependencies, yet their fixed time-window attention cannot adjust to varying traffic patterns, leading to noise accumulation. In contrast, Bi-STAT&#x002B; employs a FDETAT to separately model high-frequency fluctuations and low-frequency trends. Combined with dynamic graph structure optimization, it achieves a deep integration of spatial and temporal features, yielding superior performance.</p>
<p>To further evaluate the models&#x2019; advantages in traffic flow prediction, additional experiments were conducted on PEMS04 and PEMS08 with varying prediction horizons. Four forecast intervals were considered: short-term (1 step/5 min), medium-term (3 steps/15 min), medium-to-long-term (6 steps/30 min), and long-term (12 steps/60 min), enabling a comprehensive evaluation across time scales. Quantitative results for each model are presented in <xref ref-type="table" rid="table-3">Tables 3</xref> and <xref ref-type="table" rid="table-4">4</xref>, while <xref ref-type="fig" rid="fig-6">Fig. 6</xref> illustrates the performance trends across different prediction horizons using line charts.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Comparison of model performance on the PEMS04 dataset with different prediction steps</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Prediction Steps</th>
<th>Metric</th>
<th>HA</th>
<th>ASTGCN</th>
<th>AGCRN</th>
<th>ASTGNN</th>
<th>Bi-STAT</th>
<th>Bi-STAT&#x002B; (Ours)</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3">Length &#x003D; 1</td>
<td>MAE&#x2193;</td>
<td>26.03</td>
<td>17.93</td>
<td>18.87</td>
<td><bold>16.19</bold><sup>&#x022C6;</sup></td>
<td>17.28</td>
<td><underline>16.88</underline><sup>1</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>39.09</td>
<td>28.48</td>
<td>30.60</td>
<td><bold>26.73</bold><sup>&#x022C6;</sup></td>
<td>27.75</td>
<td><underline>27.36</underline><sup>1</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>18.90</td>
<td>12.00</td>
<td>12.71</td>
<td><bold>10.81</bold><sup>&#x022C6;</sup></td>
<td>11.71</td>
<td><underline>11.43</underline><sup>1</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 3</td>
<td>MAE&#x2193;</td>
<td>28.26</td>
<td>20.04</td>
<td>19.01</td>
<td><underline>17.73</underline><sup>1</sup></td>
<td>18.01</td>
<td><bold>17.58</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>42.33</td>
<td>31.58</td>
<td>31.17</td>
<td>29.24</td>
<td><underline>29.11</underline><sup>1</sup></td>
<td><bold>28.79</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>20.67</td>
<td>14.00</td>
<td>12.66</td>
<td><underline>11.88</underline><sup>1</sup></td>
<td>12.14</td>
<td><bold>11.87</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 6</td>
<td>MAE&#x2193;</td>
<td>31.63</td>
<td>22.07</td>
<td>19.74</td>
<td><underline>18.72</underline><sup>1</sup></td>
<td>18.77</td>
<td><bold>18.25</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>47.30</td>
<td>34.56</td>
<td>32.42</td>
<td>31.02</td>
<td><underline>30.37</underline><sup>1</sup></td>
<td><bold>30.03</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>23.42</td>
<td>15.00</td>
<td>13.06</td>
<td><underline>12.42</underline><sup>1</sup></td>
<td>12.63</td>
<td><bold>12.25</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 12</td>
<td>MAE&#x2193;</td>
<td>38.33</td>
<td>26.71</td>
<td>21.28</td>
<td>20.19</td>
<td><underline>20.17</underline><sup>1</sup></td>
<td><bold>19.36</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>57.35</td>
<td>41.10</td>
<td>34.95</td>
<td>33.32</td>
<td><underline>32.45</underline><sup>1</sup></td>
<td><bold>31.78</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>29.19</td>
<td>18.00</td>
<td>13.89</td>
<td><underline>13.29</underline><sup>1</sup></td>
<td>13.78</td>
<td><bold>12.99</bold><sup>&#x022C6;</sup></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-3fn1" fn-type="other"><p>Note: <sup>&#x022C6;</sup>Bold font indicates the best result; <sup>1</sup>Underlined font indicates the second-best result.</p></fn></table-wrap-foot>
</table-wrap><table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Comparison of model performance on the PEMS08 dataset with different prediction steps</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Prediction steps</th>
<th>Metric</th>
<th>HA</th>
<th>ASTGCN</th>
<th>AGCRN</th>
<th>ASTGNN</th>
<th>Bi-STAT</th>
<th>Bi-STAT&#x002B; (Ours)</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3">Length &#x003D; 1</td>
<td>MAE&#x2193;</td>
<td>21.31</td>
<td>14.13</td>
<td>14.43</td>
<td><underline>12.41</underline><sup>1</sup></td>
<td>13.00</td>
<td><bold>12.12</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>31.99</td>
<td>21.64</td>
<td>22.33</td>
<td><bold>19.64</bold><sup>&#x022C6;</sup></td>
<td>20.63</td>
<td><underline>19.90</underline><sup>1</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>13.37</td>
<td>9.00</td>
<td>9.57</td>
<td><underline>8.01</underline><sup>1</sup></td>
<td>8.31</td>
<td><bold>8.00</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 3</td>
<td>MAE&#x2193;</td>
<td>23.23</td>
<td>16.53</td>
<td>15.12</td>
<td>13.96</td>
<td><underline>13.40</underline><sup>1</sup></td>
<td><bold>12.71</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>34.83</td>
<td>25.49</td>
<td>23.77</td>
<td>22.41</td>
<td><underline>21.89</underline><sup>1</sup></td>
<td><bold>21.36</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>14.60</td>
<td>10.00</td>
<td>9.94</td>
<td>8.98</td>
<td><underline>8.58</underline><sup>1</sup></td>
<td><bold>8.33</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 6</td>
<td>MAE&#x2193;</td>
<td>26.12</td>
<td>18.81</td>
<td>16.20</td>
<td>15.11</td>
<td><underline>14.03</underline><sup>1</sup></td>
<td><bold>13.35</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>39.13</td>
<td>28.93</td>
<td>25.70</td>
<td>24.63</td>
<td><underline>23.34</underline><sup>1</sup></td>
<td><bold>22.80</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>16.47</td>
<td>11.00</td>
<td>10.47</td>
<td>9.77</td>
<td><underline>9.07</underline><sup>1</sup></td>
<td><bold>8.76</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 12</td>
<td>MAE&#x2193;</td>
<td>31.87</td>
<td>22.95</td>
<td>18.42</td>
<td>17.24</td>
<td><underline>15.39</underline><sup>1</sup></td>
<td><bold>14.52</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>47.76</td>
<td>34.44</td>
<td>29.13</td>
<td>28.08</td>
<td><underline>25.55</underline><sup>1</sup></td>
<td><bold>24.78</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>20.31</td>
<td>14.00</td>
<td>11.82</td>
<td>11.27</td>
<td><underline>9.99</underline><sup>1</sup></td>
<td><bold>9.63</bold><sup>&#x022C6;</sup></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-4fn1" fn-type="other"><p>Note: <sup>&#x022C6;</sup>Bold font indicates the best result; <sup>1</sup>Underlined font indicates the second-best result.</p></fn></table-wrap-foot>
</table-wrap><fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Predictive performance of PEMS dataset at different step sizes</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_69373-fig-6.tif"/>
</fig>
<p>Across different prediction horizons, model performance varies notably. Bi-STAT&#x002B; achieves the best overall results, with its advantage becoming more pronounced at longer horizons. On PEMS04, ASTGNN slightly outperforms Bi-STAT&#x002B; for short-term (1-step) predictions, while Bi-STAT&#x002B; leads in medium- to long-term forecasts (3-step, 6-step, 12-step). On PEMS08, Bi-STAT&#x002B; consistently ranks first across all horizons, with 12-step errors reduced by 0.87, 0.77, and 0.36 in the respective metrics compared with Bi-STAT. Although errors for all models increase with the prediction horizon, the growth is smallest for Bi-STAT&#x002B;.</p>
<p>The performance differences across prediction horizons stem from each model&#x2019;s capacity to capture features at different time scales. Short-term (1-step) forecasts are heavily influenced by instantaneous factors. ASTGNN&#x2019;s dynamic graph structure effectively captures local spatial correlations, whereas Bi-STAT&#x002B;&#x2019;s trend-smoothing mechanism responds slightly slower to sudden fluctuations. Long-term (12-step) forecasts depend on periodic patterns. The FDETAT module in Bi-STAT&#x002B; decomposes and accurately models low-frequency trends, thereby mitigating noise accumulation.</p>
</sec>
<sec id="s4_5_2">
<label>4.5.2</label>
<title>Comparative Experiments on Private Datasets</title>
<p>To assess the generalization capability of the proposed model, we compare the predictive performance of Bi-STAT&#x002B; with multiple benchmark models on private datasets from Baotou and Chengdu. In all experiments, the historical step size (H), current step size (P), and prediction step size (F) are set to 12. For both datasets, which have a time granularity of 1 h, the model uses the most recent 12 h of traffic data to forecast the next 12 h. Detailed results are provided in <xref ref-type="table" rid="table-5">Table 5</xref>, with all values representing the mean of 12-step forecasts.</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Performance comparison of different models on Baotou and Chengdu datasets</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Dataset</th>
<th colspan="3">Baotou</th>
<th colspan="3">Chengdu</th>
</tr>
<tr>
<th>Metric</th>
<th>MAE&#x2193;</th>
<th>RMSE&#x2193;</th>
<th>MAPE (%)&#x2193;</th>
<th>MAE&#x2193;</th>
<th>RMSE&#x2193;</th>
<th>MAPE (%)&#x2193;</th>
</tr>
</thead>
<tbody>
<tr>
<td>HA</td>
<td>250.2</td>
<td>291.48</td>
<td>296.65</td>
<td>307.56</td>
<td>345.84</td>
<td>736.15</td>
</tr>
<tr>
<td>ASTGCN</td>
<td>86.86</td>
<td>177.84</td>
<td>48.00</td>
<td>113.66</td>
<td>212.00</td>
<td>115.00</td>
</tr>
<tr>
<td>AGCRN</td>
<td>90.32</td>
<td>237.68</td>
<td>30.52</td>
<td>141.44</td>
<td>314.68</td>
<td>123.55</td>
</tr>
<tr>
<td>ASTGNN</td>
<td>74.26</td>
<td>149.49</td>
<td>40.60</td>
<td>105.83</td>
<td>204.49</td>
<td>155.86</td>
</tr>
<tr>
<td>Bi-STAT</td>
<td><underline>68.95</underline><sup>1</sup></td>
<td><underline>141.14</underline><sup>1</sup></td>
<td><underline>26.38</underline><sup>1</sup></td>
<td><underline>90.55</underline><sup>1</sup></td>
<td><underline>166.79</underline><sup>1</sup></td>
<td><underline>60.91</underline><sup>1</sup></td>
</tr>
<tr>
<td>Bi-STAT&#x002B; (ours)</td>
<td><bold>49.79</bold><sup>&#x022C6;</sup></td>
<td><bold>106.49</bold><sup>&#x022C6;</sup></td>
<td><bold>19.90</bold><sup>&#x022C6;</sup></td>
<td><bold>67.91</bold><sup>&#x022C6;</sup></td>
<td><bold>134.98</bold><sup>&#x022C6;</sup></td>
<td><bold>54.50</bold><sup>&#x022C6;</sup></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-5fn1" fn-type="other"><p>Note: <sup>&#x022C6;</sup>3Bold font indicates the best result; <sup>1</sup>Underlined font indicates the second-best result.</p></fn></table-wrap-foot>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="table-5">Table 5</xref>, Bi-STAT&#x002B; achieves the best performance across all evaluation metrics for both Baotou and Chengdu. These results indicate that Bi-STAT&#x002B; maintains stable and high predictive accuracy despite the spatio-temporal variations in traffic patterns between cities, thereby demonstrating strong generalization capability.</p>

<p>Similar to the experiments on the public datasets, we also conducted comparative experiments with varying prediction horizons on the Baotou and Chengdu private datasets. The quantitative results for each model are presented in <xref ref-type="table" rid="table-6">Tables 6</xref> and <xref ref-type="table" rid="table-7">7</xref>, while <xref ref-type="fig" rid="fig-7">Fig. 7</xref> provides a line chart illustrating how prediction performance changes with the forecast horizon.</p>
<table-wrap id="table-6">
<label>Table 6</label>
<caption>
<title>Comparison of model performance with different prediction steps on the Baotou dataset</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Prediction steps</th>
<th>Metric</th>
<th>HA</th>
<th>ASTGCN</th>
<th>AGCRN</th>
<th>ASTGNN</th>
<th>Bi-STAT</th>
<th>Bi-STAT&#x002B; (Ours)</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3">Length &#x003D; 1</td>
<td>MAE&#x2193;</td>
<td>431.59</td>
<td>75.33</td>
<td>90.44</td>
<td><underline>62.61</underline><sup>1</sup></td>
<td>65.69</td>
<td><bold>43.40</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>573.35</td>
<td>152.74</td>
<td>241.65</td>
<td><underline>119.04</underline><sup>1</sup></td>
<td>132.98</td>
<td><bold>90.32</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>46.70</td>
<td>42.00</td>
<td>35.92</td>
<td>29.96</td>
<td><underline>26.53</underline><sup>1</sup></td>
<td><bold>18.81</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 3</td>
<td>MAE&#x2193;</td>
<td>67.40</td>
<td>86.72</td>
<td>90.55</td>
<td>73.24</td>
<td><underline>67.36</underline><sup>1</sup></td>
<td><bold>47.59</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>114.06</td>
<td>173.52</td>
<td>232.91</td>
<td>144.18</td>
<td><underline>137.39</underline><sup>1</sup></td>
<td><bold>102.56</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td><bold>16.18</bold><sup>&#x022C6;</sup></td>
<td>47.00</td>
<td>31.69</td>
<td>39.13</td>
<td>26.25</td>
<td><underline>20.10</underline><sup>1</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 6</td>
<td>MAE&#x2193;</td>
<td>261.79</td>
<td>85.43</td>
<td>88.86</td>
<td>73.79</td>
<td><underline>66.90</underline><sup>1</sup></td>
<td><bold>49.56</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>315.95</td>
<td>177.14</td>
<td>232.18</td>
<td>146.31</td>
<td><underline>138.02</underline><sup>1</sup></td>
<td><bold>105.67</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>184.86</td>
<td>45.00</td>
<td>29.62</td>
<td>41.27</td>
<td><underline>26.17</underline><sup>1</sup></td>
<td><bold>19.74</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 12</td>
<td>MAE&#x2193;</td>
<td>350.53</td>
<td>87.45</td>
<td>95.07</td>
<td>78.07</td>
<td><underline>71.86</underline><sup>1</sup></td>
<td><bold>53.31</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>421.91</td>
<td>180.58</td>
<td>256.33</td>
<td>164.71</td>
<td><underline>146.67</underline><sup>1</sup></td>
<td><bold>113.92</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>471.28</td>
<td>46.00</td>
<td>29.42</td>
<td>38.77</td>
<td><underline>26.44</underline><sup>1</sup></td>
<td><bold>20.01</bold><sup>&#x022C6;</sup></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-6fn1" fn-type="other"><p>Note: <sup>&#x022C6;</sup>Bold font indicates the best result; <sup>1</sup>Underlined font indicates the second-best result.</p></fn></table-wrap-foot>
</table-wrap><table-wrap id="table-7">
<label>Table 7</label>
<caption>
<title>Comparison of model performance with different prediction steps on the Chengdu dataset</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Prediction steps</th>
<th>Metric</th>
<th>HA</th>
<th>ASTGCN</th>
<th>AGCRN</th>
<th>ASTGNN</th>
<th>Bi-STAT</th>
<th>Bi-STAT&#x002B; (Ours)</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3">Length &#x003D; 1</td>
<td>MAE&#x2193;</td>
<td>193.46</td>
<td>85.25</td>
<td>159.53</td>
<td><underline>74.63</underline><sup>1</sup></td>
<td>76.00</td>
<td><bold>61.81</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>254.37</td>
<td>165.65</td>
<td>340.46</td>
<td><underline>142.43</underline><sup>1</sup></td>
<td>144.64</td>
<td><bold>122.13</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>572.01</td>
<td>79.00</td>
<td>234.92</td>
<td>57.22</td>
<td><bold>40.21</bold><sup>&#x022C6;</sup></td>
<td><underline>46.93</underline><sup>1</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 3</td>
<td>MAE&#x2193;</td>
<td>338.64</td>
<td>117.59</td>
<td>135.31</td>
<td>104.66</td>
<td><underline>86.94</underline><sup>1</sup></td>
<td><bold>63.48</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>439.46</td>
<td>211.56</td>
<td>292.20</td>
<td>198.75</td>
<td><underline>163.90</underline><sup>1</sup></td>
<td><bold>127.92</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>1793.78</td>
<td>140.00</td>
<td>127.39</td>
<td>147.70</td>
<td><bold>51.26</bold><sup>&#x022C6;</sup></td>
<td><underline>52.51</underline><sup>1</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 6</td>
<td>MAE&#x2193;</td>
<td>431.91</td>
<td>114.75</td>
<td>134.72</td>
<td>107.53</td>
<td><underline>95.31</underline><sup>1</sup></td>
<td><bold>67.33</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>561.40</td>
<td>210.14</td>
<td>296.60</td>
<td>202.87</td>
<td><underline>173.19</underline><sup>1</sup></td>
<td><bold>134.77</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>3539.31</td>
<td>130.00</td>
<td>118.34</td>
<td>187.49</td>
<td><underline>65.33</underline><sup>1</sup></td>
<td><bold>56.31</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">Length &#x003D; 12</td>
<td>MAE&#x2193;</td>
<td>298.81</td>
<td>109.84</td>
<td>172.74</td>
<td>107.48</td>
<td><underline>86.52</underline><sup>1</sup></td>
<td><bold>74.78</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>419.36</td>
<td>212.14</td>
<td>384.38</td>
<td>211.34</td>
<td><underline>159.48</underline><sup>1</sup></td>
<td><bold>145.22</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>140.36</td>
<td>89.00</td>
<td>97.89</td>
<td>85.05</td>
<td><underline>62.66</underline><sup>1</sup></td>
<td><bold>56.59</bold><sup>&#x022C6;</sup></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-7fn1" fn-type="other"><p>Note: <sup>&#x022C6;</sup>Bold font indicates the best result; <sup>1</sup>Underlined font indicates the second-best result.</p></fn></table-wrap-foot>
</table-wrap><fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Predictive performance of Baotou and Chengdu datasets at different step sizes</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_69373-fig-7.tif"/>
</fig>
</sec>
</sec>
<sec id="s4_6">
<label>4.6</label>
<title>Ablation Study</title>
<p>This study conducts ablation experiments on four datasets to evaluate the contribution of each core innovation in the Bi-STAT&#x002B; model to traffic flow prediction. Four variant models are designed to analyze the specific effects of individual components.</p>
<p>Variant Model 1: Removes the FECAM module and replaces the Agent Attention Mechanism with multi-head attention to evaluate the combined contribution of FECAM and Agent Attention to overall performance.</p>
<p>Variant Model 2: Removes the FECAM module from FDETAT, which reduces the model&#x2019;s ability to capture temporal contextual information.</p>
<p>Variant Model 3: Replaces the Agent Attention Mechanism in SAAT with multi-head attention, thereby reducing the model&#x2019;s ability to adaptively allocate attention across nodes.</p>
<p>Variant Model 4: Removes the Adaptive Reconciliation Embedding (AAE) module, reducing the model&#x2019;s ability to embed dynamic temporal and spatial information.</p>
<p><xref ref-type="table" rid="table-8">Table 8</xref> summarizes the performance of Bi-STAT&#x002B; and its variants across the four datasets, followed by the corresponding analysis.</p>

<p><list list-type="order">
<list-item>
<p>Contribution of the FECAM Module: Removing the FECAM module led to a notable performance decline, particularly on the Baotou dataset, where the MAE rose from 49.79 to 54.80. This demonstrates that the FECAM module leverages frequency-domain analysis to jointly model high-frequency details and low-frequency trends, thereby improving the representation of multi-scale temporal features.</p></list-item>
<list-item>
<p>Effectiveness of the Agent Attention Mechanism: Replacing the Agent Attention Mechanism with traditional multi-head attention resulted in a clear drop in prediction accuracy across all four datasets. For example, on the PEMS04 dataset, the MAE increased from 18.33 to 18.72. This finding indicates that the Agent Attention Mechanism adaptively allocates attention weights to traffic sensors at different locations based on dynamic traffic network changes, enabling better capture of both local and global spatial features. Its advantage is particularly evident on the PEMS04 dataset, which features a large number of nodes and a complex topology.</p></list-item>
<list-item>
<p>Importance of the AAE Module: Removing the AAE module also reduced performance. On the PEMS04 dataset, the MAE increased from 18.33 to 18.62. The AAE module strengthens spatio-temporal feature embedding through adaptive reconciliation matrices, allowing Bi-STAT&#x002B; to better capture dynamic spatio-temporal dependencies and thereby enhance prediction accuracy.</p></list-item>
</list></p>
<table-wrap id="table-8">
<label>Table 8</label>
<caption>
<title>Predictive performance of Bi-STAT&#x002B; and variants on four datasets</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Dataset</th>
<th>Metric</th>
<th>Variant model 1</th>
<th>Variant model 2</th>
<th>Variant model 3</th>
<th>Variant model 4</th>
<th>Full model</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3">PEMS04</td>
<td>MAE&#x2193;</td>
<td>18.66</td>
<td>18.43</td>
<td>18.72</td>
<td>18.62</td>
<td>18.33</td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>30.15</td>
<td>30.01</td>
<td>30.21</td>
<td>30.24</td>
<td>29.98</td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>12.53</td>
<td>12.59</td>
<td>12.88</td>
<td>12.60</td>
<td>12.21</td>
</tr>
<tr>
<td rowspan="3">PEMS08</td>
<td>MAE&#x2193;</td>
<td>13.57</td>
<td>13.40</td>
<td>13.51</td>
<td>13.79</td>
<td>13.39</td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>22.95</td>
<td>23.02</td>
<td>22.80</td>
<td>23.00</td>
<td>22.74</td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>9.08</td>
<td>8.98</td>
<td>8.93</td>
<td>9.02</td>
<td>8.82</td>
</tr>
<tr>
<td rowspan="3">Baotou</td>
<td>MAE&#x2193;</td>
<td>65.28</td>
<td>54.8</td>
<td>63.32</td>
<td>54.96</td>
<td>49.79</td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>127.46</td>
<td>113.68</td>
<td>126.93</td>
<td>112.21</td>
<td>106.49</td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>25.61</td>
<td>22.14</td>
<td>25.16</td>
<td>21.04</td>
<td>19.90</td>
</tr>
<tr>
<td rowspan="3">Chengdu</td>
<td>MAE&#x2193;</td>
<td>80.89</td>
<td>71.81</td>
<td>83.48</td>
<td>74.65</td>
<td>67.91</td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>149.41</td>
<td>138.52</td>
<td>153.66</td>
<td>141.65</td>
<td>134.98</td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>63.76</td>
<td>59.26</td>
<td>62.03</td>
<td>56.23</td>
<td>54.50</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_7">
<label>4.7</label>
<title>Robustness Analysis</title>
<p>In real-world traffic flow data collection, sensor measurements are frequently affected by factors such as poor contact and natural aging, resulting in noise interference. Environmental conditions and limited sensor lifespans make it difficult to fully eliminate noise caused by aging. Such interference can significantly reduce the accuracy of traffic flow prediction models.</p>
<p>To evaluate the robustness of Bi-STAT&#x002B; in noisy environments, we compare its performance with several benchmark models on the Baotou dataset. Data loss is simulated by randomly removing 20%, 40%, and 60% of the records. Gaussian noise with a mean of 10 and a standard deviation of 500 is then added at the same proportions to mimic real-world uncertainties, including measurement errors and outliers. <xref ref-type="table" rid="table-9">Tables 9</xref> and <xref ref-type="table" rid="table-10">10</xref> summarize the prediction performance of Bi-STAT&#x002B; and the benchmark models under different levels of missing data and noise.</p>
<table-wrap id="table-9">
<label>Table 9</label>
<caption>
<title>Robustness experiments of the Baotou dataset with different scales of deletion</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Rate</th>
<th>Metric</th>
<th>HA</th>
<th>ASTGCN</th>
<th>AGCRN</th>
<th>ASTGNN</th>
<th>Bi-STAT</th>
<th>Bi-STAT&#x002B; (Ours)</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3">0%</td>
<td>MAE&#x2193;</td>
<td>250.2</td>
<td>86.86</td>
<td>90.31</td>
<td>74.26</td>
<td><underline>68.95</underline><sup>1</sup></td>
<td><bold>49.79</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>291.48</td>
<td>177.84</td>
<td>237.68</td>
<td>149.49</td>
<td><underline>141.14</underline><sup>1</sup></td>
<td><bold>106.49</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>296.65</td>
<td>48.00</td>
<td>30.52</td>
<td>40.60</td>
<td><underline>26.38</underline><sup>1</sup></td>
<td><bold>19.90</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">20%</td>
<td>MAE&#x2193;</td>
<td>286.36</td>
<td>171.13</td>
<td>184.74</td>
<td>138.87</td>
<td><underline>70.86</underline><sup>1</sup></td>
<td><bold>52.34</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>218.9</td>
<td>321.22</td>
<td>359.2</td>
<td>303.28</td>
<td><underline>138.41</underline><sup>1</sup></td>
<td><bold>116.36</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>246.19</td>
<td>54.00</td>
<td>35.58</td>
<td>24.25</td>
<td><underline>28.30</underline><sup>1</sup></td>
<td><bold>20.93</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">40%</td>
<td>MAE&#x2193;</td>
<td>285.7</td>
<td>240.89</td>
<td>228.57</td>
<td>219.09</td>
<td><underline>75.68</underline><sup>1</sup></td>
<td><bold>56.41</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>319.02</td>
<td>395.39</td>
<td>383.77</td>
<td>384.01</td>
<td><underline>159.1</underline><sup>1</sup></td>
<td><bold>136.46</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>187.26</td>
<td>66.00</td>
<td>371,311,872.00</td>
<td>37.05</td>
<td><underline>28.78</underline><sup>1</sup></td>
<td><bold>21.53</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">60%</td>
<td>MAE&#x2193;</td>
<td>240.59</td>
<td>183.68</td>
<td>186.67</td>
<td>183.88</td>
<td><underline>76.45</underline><sup>1</sup></td>
<td><bold>57.15</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>290.46</td>
<td>418.17</td>
<td>422.41</td>
<td>417.81</td>
<td><underline>159.88</underline><sup>1</sup></td>
<td><bold>139.49</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>138.32</td>
<td>100.00</td>
<td>100.00</td>
<td>99.32</td>
<td><underline>30.73</underline><sup>1</sup></td>
<td><bold>22.18</bold><sup>&#x022C6;</sup></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-9fn1" fn-type="other"><p>Note: <sup>&#x022C6;</sup>Bold font indicates the best result; <sup>1</sup>Underlined font indicates the second-best result.</p></fn></table-wrap-foot>
</table-wrap><table-wrap id="table-10">
<label>Table 10</label>
<caption>
<title>Robustness experiments of the Baotou dataset with different proportions of noise</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Rate</th>
<th>Metric</th>
<th>HA</th>
<th>ASTGCN</th>
<th>AGCRN</th>
<th>ASTGNN</th>
<th>Bi-STAT</th>
<th>Bi-STAT&#x002B;<break/> (Ours)</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3">0%</td>
<td>MAE&#x2193;</td>
<td>250.2</td>
<td>86.86</td>
<td>90.31</td>
<td>74.26</td>
<td><underline>68.95</underline><sup>1</sup></td>
<td><bold>49.79</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>291.48</td>
<td>177.84</td>
<td>237.68</td>
<td>149.49</td>
<td><underline>141.14</underline><sup>1</sup></td>
<td><bold>106.49</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>296.65</td>
<td>48.00</td>
<td>30.52</td>
<td>40.60</td>
<td><underline>26.38</underline><sup>1</sup></td>
<td><bold>19.90</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">20%</td>
<td>MAE&#x2193;</td>
<td>297.2</td>
<td>184.7</td>
<td>168.67</td>
<td><underline>132.93</underline><sup>1</sup></td>
<td>142.7</td>
<td><bold>126.18</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>382.43</td>
<td>314.04</td>
<td>339.00</td>
<td><underline>263.44</underline><sup>1</sup></td>
<td>271.87</td>
<td><bold>255.31</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>292.35</td>
<td>100.00</td>
<td>49.55</td>
<td>18,400.57</td>
<td><underline>20.95</underline><sup>1</sup></td>
<td><bold>14.99</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">40%</td>
<td>MAE&#x2193;</td>
<td>340.61</td>
<td>261.38</td>
<td>248.75</td>
<td><underline>200.67</underline><sup>1</sup></td>
<td>211.05</td>
<td><bold>195.28</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>444.99</td>
<td>400.51</td>
<td>427.74</td>
<td><underline>341.68</underline><sup>1</sup></td>
<td>349.28</td>
<td><bold>339.26</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>294.46</td>
<td>154.00</td>
<td>80.31</td>
<td>102.99</td>
<td><underline>17.61</underline><sup>1</sup></td>
<td><bold>12.32</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">60%</td>
<td>MAE&#x2193;</td>
<td>386.15</td>
<td>340.34</td>
<td>291.51</td>
<td><underline>271.78</underline><sup>1</sup></td>
<td>277.28</td>
<td><bold>267.18</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>501.73</td>
<td>480.49</td>
<td>440.45</td>
<td><underline>409.83</underline><sup>1</sup></td>
<td>413.11</td>
<td><bold>406.33</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>291.91</td>
<td>202.00</td>
<td>639,995.44</td>
<td>143.15</td>
<td><underline>21.59</underline><sup>1</sup></td>
<td><bold>16.92</bold><sup>&#x022C6;</sup></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-10fn1" fn-type="other"><p>Note: <sup>&#x022C6;</sup>Bold font indicates the best result; <sup>1</sup>Underlined font indicates the second-best result.</p></fn></table-wrap-foot>
</table-wrap>
<p>The results show that Bi-STAT&#x002B; exhibits greater robustness than the benchmark models under missing data and noise conditions. With 20% missing data and noise, the MAE of Bi-STAT&#x002B; rises only slightly, while other graph-based models suffer more substantial degradation. Even as missing data and noise levels reach 40% and 60%, Bi-STAT&#x002B; maintains relatively stable accuracy, with only moderate increases in MAE. Under severe noise interference, it effectively suppresses the impact of Gaussian noise, demonstrating strong noise tolerance.</p>
</sec>
<sec id="s4_8">
<label>4.8</label>
<title>Sampling Frequency Analysis</title>
<p>As shown in <xref ref-type="sec" rid="s4_5">Section 4.5</xref>, prediction errors on private datasets are notably higher than on public datasets, mainly due to differences in data collection frequency. The PEMS dataset employs a 5-min sampling interval, which yields strong correlations between consecutive time steps. Consequently, a 12-step prediction covers only the next hour, thereby facilitating the capture of short-term dynamics. In contrast, private datasets are sampled hourly, greatly weakening temporal correlations. A 12-step prediction must then span 12 h of traffic flow changes, thereby amplifying cumulative errors and increasing the difficulty of the prediction.</p>
<p>To examine the effect of sampling frequency on prediction accuracy, we performed resampling experiments on the Baotou and Chengdu datasets. The original traffic data were resampled to 30-min, 15, 10, and 5-min intervals. Two preprocessing steps were applied: (1) linear interpolation filled in missing time steps during resampling; and (2) a moving-average method smoothed the prediction values for the final time step. <xref ref-type="table" rid="table-11">Tables 11</xref> and <xref ref-type="table" rid="table-12">12</xref> present the prediction errors of Bi-STAT&#x002B; and the benchmark models under different sampling frequencies.</p>
<table-wrap id="table-11">
<label>Table 11</label>
<caption>
<title>Predictive performance of the model on the Baotou dataset at different data collection frequencies</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Period</th>
<th>Metric</th>
<th>HA</th>
<th>ASTGCN</th>
<th>AGCRN</th>
<th>ASTGNN</th>
<th>Bi-STAT</th>
<th>Bi-STAT&#x002B; (Ours)</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3">30 min</td>
<td>MAE&#x2193;</td>
<td>245.49</td>
<td>80.12</td>
<td>74.66</td>
<td>68.04</td>
<td><underline>62.89</underline><sup>1</sup></td>
<td><bold>45.01</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>284.75</td>
<td>166.41</td>
<td>214.94</td>
<td>149.58</td>
<td><underline>124.4</underline><sup>1</sup></td>
<td><bold>100.69</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>267.82</td>
<td>37.00</td>
<td><underline>21.00</underline><sup>1</sup></td>
<td>33.89</td>
<td>23.49</td>
<td><bold>14.84</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">15 min</td>
<td>MAE&#x2193;</td>
<td>241.78</td>
<td>72.60</td>
<td>48.32</td>
<td><underline>43.98</underline><sup>1</sup></td>
<td>49.29</td>
<td><bold>35.57</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>281.07</td>
<td>146.66</td>
<td>147.24</td>
<td>96.15</td>
<td><underline>90.86</underline><sup>1</sup></td>
<td><bold>80.96</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>252.77</td>
<td>26.00</td>
<td><underline>14.31</underline><sup>1</sup></td>
<td>15.28</td>
<td>17.66</td>
<td><bold>12.05</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">10 min</td>
<td>MAE&#x2193;</td>
<td>240.98</td>
<td>49.61</td>
<td>38.78</td>
<td><underline>33.33</underline><sup>1</sup></td>
<td>41.62</td>
<td><bold>27.86</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>280.21</td>
<td>120.23</td>
<td>123.75</td>
<td>76.49</td>
<td><underline>75.70</underline><sup>1</sup></td>
<td><bold>59.04</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>249.39</td>
<td>16.00</td>
<td><underline>12.08</underline><sup>1</sup></td>
<td>13.24</td>
<td>15.08</td>
<td><bold>9.57</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">5 min</td>
<td>MAE&#x2193;</td>
<td>240.84</td>
<td>36.79</td>
<td>24.05</td>
<td>22.4</td>
<td><underline>20.39</underline><sup>1</sup></td>
<td><bold>20.06</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>280.02</td>
<td>87.14</td>
<td>108.17</td>
<td>55.52</td>
<td><underline>40.57</underline><sup>1</sup></td>
<td><bold>39.85</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>247.55</td>
<td>12.00</td>
<td><bold>6.49</bold><sup>&#x022C6;</sup></td>
<td>8.40</td>
<td>8.45</td>
<td><underline>6.82</underline><sup>1</sup></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-11fn1" fn-type="other"><p>Note: <sup>&#x022C6;</sup>Bold font indicates the best result; <sup>1</sup>Underlined font indicates the second-best result.</p></fn></table-wrap-foot>
</table-wrap><table-wrap id="table-12">
<label>Table 12</label>
<caption>
<title>Predictive performance of the model on the Chengdu dataset at different data collection frequencies</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Period</th>
<th>Metric</th>
<th>HA</th>
<th>ASTGCN</th>
<th>AGCRN</th>
<th>ASTGNN</th>
<th>Bi-STAT</th>
<th>Bi-STAT&#x002B; (Ours)</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3">30 min</td>
<td>MAE&#x2193;</td>
<td>305.58</td>
<td>87.26</td>
<td>78.49</td>
<td><underline>69.61</underline><sup>1</sup></td>
<td>72.34</td>
<td><bold>64.26</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>339.70</td>
<td>163.01</td>
<td>167.21</td>
<td>148.36</td>
<td><underline>136.01</underline><sup>1</sup></td>
<td><bold>126.34</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>674.23</td>
<td>77.00</td>
<td>69.22</td>
<td>51.79</td>
<td><underline>46.40</underline><sup>1</sup></td>
<td><bold>41.87</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">15 min</td>
<td>MAE&#x2193;</td>
<td>301.47</td>
<td>68.32</td>
<td><underline>59.57</underline><sup>1</sup></td>
<td>64.78</td>
<td>68.60</td>
<td><bold>51.79</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>338.04</td>
<td>136.96</td>
<td>130.59</td>
<td>140.66</td>
<td><underline>127.22</underline><sup>1</sup></td>
<td><bold>96.76</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>646.99</td>
<td>44.00</td>
<td>37.86</td>
<td>35.12</td>
<td><underline>32.02</underline><sup>1</sup></td>
<td><bold>28.18</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">10 min</td>
<td>MAE&#x2193;</td>
<td>301.9</td>
<td>60.75</td>
<td><underline>51.03</underline><sup>1</sup></td>
<td>57.96</td>
<td>63.55</td>
<td><bold>46.74</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>338.43</td>
<td>126.13</td>
<td><underline>117.66</underline><sup>1</sup></td>
<td>122.97</td>
<td>119.37</td>
<td><bold>88.9</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>641.02</td>
<td>29.00</td>
<td><underline>26.00</underline><sup>1</sup></td>
<td>29.37</td>
<td>27.15</td>
<td><bold>23.29</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td rowspan="3">5 min</td>
<td>MAE&#x2193;</td>
<td>301.46</td>
<td>34.33</td>
<td>18.31</td>
<td><underline>17.81</underline><sup>1</sup></td>
<td>24.81</td>
<td><bold>16.25</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>RMSE&#x2193;</td>
<td>337.90</td>
<td>73.78</td>
<td><underline>44.66</underline><sup>1</sup></td>
<td>47.6</td>
<td>52.4</td>
<td><bold>36.49</bold><sup>&#x022C6;</sup></td>
</tr>
<tr>
<td>MAPE (%)&#x2193;</td>
<td>633.54</td>
<td>22.00</td>
<td>10.45</td>
<td><underline>9.02</underline><sup>1</sup></td>
<td>10.94</td>
<td><bold>7.11</bold><sup>&#x022C6;</sup></td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-12fn1" fn-type="other"><p>Note: <sup>&#x022C6;</sup>Bold font indicates the best result; <sup>1</sup>Underlined font indicates the second-best result.</p></fn></table-wrap-foot>
</table-wrap>
<p>Experimental results show that as the sampling interval decreases from 1 h to 5 min, the prediction errors of both Bi-STAT&#x002B; and the benchmark model decline markedly. This supports our hypothesis that longer intervals increase prediction uncertainty and error. Shorter intervals strengthen temporal correlations, allowing the model to capture traffic flow dynamics more accurately and thereby reduce prediction errors.</p>
</sec>
<sec id="s4_9">
<label>4.9</label>
<title>Visual Result Analysis</title>
<p>Using the 5-min sampling frequency traffic flow dataset from Baotou City, two sensor nodes with distinct location characteristics&#x2014;No. 98 (residential area) and No. 205 (commercial area)&#x2014;were selected for analysis, with their spatial positions shown in <xref ref-type="fig" rid="fig-8">Fig. 8a</xref>,<xref ref-type="fig" rid="fig-8">b</xref>. Traffic flow prediction performance was evaluated for different forecast horizons (15, 30, and 60 min) on 07 July 2024 (Sunday). As shown in <xref ref-type="fig" rid="fig-9">Fig. 9a</xref>&#x2013;<xref ref-type="fig" rid="fig-9">f</xref>, the Bi-STAT&#x002B; predictions (red dashed line) closely align with the actual traffic flow (blue solid line) over time. Across both short-term (15 min) and long-term (60 min) forecasts, Bi-STAT&#x002B; maintains high accuracy and stability across different nodes. Notably, during peak traffic periods, it captures rapid fluctuations in traffic speed and accurately reflects dynamic changes in traffic volume. Moreover, performance variations between nodes in different locations suggest that Bi-STAT&#x002B; offers advantages in handling geospatial heterogeneity.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>(<bold>a</bold>) Node 98 (red marker); (<bold>b</bold>) Node 205 (red marker)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_69373-fig-8.tif"/>
</fig><fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>(<bold>a</bold>) Plot of sensor node #98 at 15 min of prediction; (<bold>b</bold>) Plot of sensor node #98 at 30 min of prediction; (<bold>c</bold>) Plot of sensor node #98 at 60 min of prediction; (<bold>d</bold>) Plot of sensor node #205 at 15 min of prediction; (<bold>e</bold>) Plot of sensor node #205 at 30 min of prediction; (<bold>f</bold>) Plot of sensor node #205 at 60 min of prediction. (blue straight line is true value, red dashed line is predicted value)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_69373-fig-9.tif"/>
</fig>
</sec>
<sec id="s4_10">
<label>4.10</label>
<title>Real-Time Applicability Evaluation</title>
<p>To validate the real-time capability of Bi-STAT&#x002B;, we measured its inference speed on traffic networks from Baotou and Chengdu under identical experimental settings (<xref ref-type="sec" rid="s4_3">Section 4.3</xref>). Results show that Bi-STAT&#x002B; processes full-network predictions in 0.97 s (Baotou) and 3.88 s (Chengdu), while the baseline Bi-STAT requires 0.15 and 1.52 s, respectively. Although Bi-STAT&#x002B; incurs higher computational costs due to its enhanced architecture, both models operate well within the 5 s real-time threshold for traffic management systems. This efficiency ensures practical deployment in smart city applications without sacrificing prediction quality.</p>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusions</title>
<p>This study presents Bi-STAT&#x002B;, an advanced spatio-temporal transformer framework that significantly advances urban traffic flow forecasting. The proposed model integrates dynamic adaptive encoding via STAE with enhanced attention mechanisms (FDETAT/SAAT), demonstrating superior capability in capturing complex spatio-temporal dependencies. Extensive experimental evaluations on real-world datasets validate the framework&#x2019;s effectiveness, showing consistent improvements of 15.6% in prediction accuracy and 11.6% in robustness compared to SOTA methods. The systematic analysis of sampling frequency effects provides valuable insights for practical implementation. These contributions not only establish a new benchmark in traffic forecasting research but also offer significant potential for real-world applications in intelligent transportation systems and smart city development.</p>
<p>Future research will proceed in two directions. First, we aim to further improve the model&#x2019;s computational efficiency. Although the current version meets real-time application requirements, ongoing urbanization and rising vehicle ownership continue to increase the complexity and scale of traffic data. This creates opportunities to optimize the model&#x2019;s architecture and reduce inference time to meet stricter performance demands. Second, we will address the challenge of predicting traffic flow at checkpoints lacking historical data. By integrating external information such as road network topology and POI distribution, we plan to develop an adaptive framework that leverages data from adjacent checkpoints to estimate traffic flow at newly deployed nodes, thereby broadening the model&#x2019;s applicability.</p>
</sec>
</body>
<back>
<ack>
<p>Not applicable.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>This work was partly supported by the Youth Foundation of the Inner Mongolia Natural Science Foundation [grant number 2024QN06017 and 2025MS06022], the Basic Scientific Research Business Fee Project for Universities in Inner Mongolia [grant numbers 2023XKJX019 and 2023XKJX024], the Central Guidance on Local Science and Technology Development Fund through [grant number 2024ZY0084].</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>Conceptualization, Yali Cao and Weijian Hu; methodology, Yali Cao; software, Yali Cao; validation, Yali Cao, Weijian Hu and Lingfang Li; formal analysis, Yali Cao; investigation, Yali Cao; resources, Yali Cao; data curation, Yali Cao; writing&#x2014;original draft preparation, Yali Cao; writing&#x2014;review and editing, Yali Cao, Weijian Hu, Lingfang Li, Minchao Li, Meng Xu and Ke Han; visualization, Yali Cao; supervision, Weijian Hu, Lingfang Li, Minchao Li, Meng Xu and Ke Han; project administration, Yali Cao; funding acquisition, Weijian Hu. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>Data available on request from the authors. The data that support the findings of this study are available from the corresponding author, Lingfang Li, upon reasonable request.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<glossary content-type="abbreviations" id="glossary-1">
<title>Abbreviations</title>
<def-list>
<def-item>
<term>ITS</term>
<def>
<p>Intelligent transportation systems</p>
</def>
</def-item>
<def-item>
<term>CNN</term>
<def>
<p>Convolutional neural network</p>
</def>
</def-item>
<def-item>
<term>GNN</term>
<def>
<p>Graph neural network</p>
</def>
</def-item>
<def-item>
<term>GCN</term>
<def>
<p>Graph convolution network</p>
</def>
</def-item>
<def-item>
<term>RNN</term>
<def>
<p>Recurrent neural network</p>
</def>
</def-item>
<def-item>
<term>LSTM</term>
<def>
<p>Long short-term memory</p>
</def>
</def-item>
<def-item>
<term>GRU</term>
<def>
<p>Gated recurrent unit</p>
</def>
</def-item>
<def-item>
<term>STGCN</term>
<def>
<p>Spatio-temporal graph convolution</p>
</def>
</def-item>
<def-item>
<term>TCN</term>
<def>
<p>Temporal convolutional network</p>
</def>
</def-item>
<def-item>
<term>MLP</term>
<def>
<p>Multilayer perceptron</p>
</def>
</def-item>
<def-item>
<term>DHM</term>
<def>
<p>Dynamic halting module</p>
</def>
</def-item>
<def-item>
<term>DCT</term>
<def>
<p>Discrete cosine transform</p>
</def>
</def-item>
<def-item>
<term>HA</term>
<def>
<p>Historical average</p>
</def>
</def-item>
<def-item>
<term>DWC</term>
<def>
<p>Depthwise separable convolution</p>
</def>
</def-item>
<def-item>
<term>PEMS</term>
<def>
<p>Performance measurement system</p>
</def>
</def-item>
<def-item>
<term>MAE</term>
<def>
<p>Mean absolute error</p>
</def>
</def-item>
<def-item>
<term>RMSE</term>
<def>
<p>Root mean squared error</p>
</def>
</def-item>
<def-item>
<term>MAPE</term>
<def>
<p>Mean absolute percentage error</p>
</def>
</def-item>
<def-item>
<term>TCN</term>
<def>
<p>Temporal convolutional networks</p>
</def>
</def-item>
</def-list>
</glossary>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhou</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>C</given-names></string-name>, <string-name><surname>Song</surname> <given-names>C</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>X</given-names></string-name>, <string-name><surname>Chang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Short-term traffic flow prediction of the smart city using 5G Internet of vehicles based on edge computing</article-title>. <source>IEEE Trans Intell Transp Syst</source>. <year>2023</year>;<volume>24</volume>(<issue>2</issue>):<fpage>2229</fpage>&#x2013;<lpage>38</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TITS.2022.3147845</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>L</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>X</given-names></string-name></person-group>. <article-title>Traffic flow matrix-based graph neural network with attention mechanism for traffic flow prediction</article-title>. <source>Inf Fusion</source>. <year>2024</year>;<volume>104</volume>(<issue>6</issue>):<fpage>102146</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.inffus.2023.102146</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Patras</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Long-term mobile traffic forecasting using deep spatio-temporal neural networks</article-title>. In: <conf-name>Proceedings of the Eighteenth ACM International Symposium on Mobile Ad Hoc Networking and Computing; 2018 Jun 26&#x2013;29</conf-name>; <publisher-loc>Los Angeles, CA, USA</publisher-loc>. doi:<pub-id pub-id-type="doi">10.1145/3209582.3209606</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Shao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>F</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Fang</surname> <given-names>J</given-names></string-name></person-group>. <article-title>The traffic flow prediction method using the incremental learning-based CNN-LTSM model: the solution of mobile application</article-title>. <source>Mob Inf Syst</source>. <year>2021</year>;<volume>2021</volume>(<issue>4</issue>):<fpage>5579451</fpage>. doi:<pub-id pub-id-type="doi">10.1155/2021/5579451</pub-id>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yuan</surname> <given-names>H</given-names></string-name>, <string-name><surname>Li</surname> <given-names>G</given-names></string-name>, <string-name><surname>Bao</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Feng</surname> <given-names>L</given-names></string-name></person-group>. <article-title>An effective joint prediction model for travel demands and traffic flows</article-title>. In: <conf-name>2021 IEEE 37th International Conference on Data Engineering (ICDE); 2021 Apr 19&#x2013;22; Chania, Greece</conf-name>. doi:<pub-id pub-id-type="doi">10.1109/icde51399.2021.00037</pub-id>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Qi</surname> <given-names>X</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>W</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Han</surname> <given-names>K</given-names></string-name></person-group>. <article-title>STGNN-FAM: a traffic flow prediction model for spatiotemporal graph networks based on fusion of attention mechanisms</article-title>. <source>J Adv Transp</source>. <year>2023</year>;<volume>2023</volume>(<issue>2</issue>):<fpage>8880530</fpage>. doi:<pub-id pub-id-type="doi">10.1155/2023/8880530</pub-id>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Bai</surname> <given-names>L</given-names></string-name>, <string-name><surname>Yao</surname> <given-names>L</given-names></string-name>, <string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Adaptive graph convolutional recurrent network for traffic forecasting</article-title>. In: <conf-name>Proceedings of the 34th Conference on Neural Information Processing Systems (NeurIPS 2020); 2020 Dec 6&#x2013;12</conf-name>; <publisher-loc>Vancouver, BC, Canada</publisher-loc>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Guo</surname> <given-names>S</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Feng</surname> <given-names>N</given-names></string-name>, <string-name><surname>Song</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Attention based spatial-temporal graph convolutional networks for traffic flow forecasting</article-title>. <source>Proc AAAI Conf Artif Intell</source>. <year>2019</year>;<volume>33</volume>(<issue>1</issue>):<fpage>922</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1609/aaai.v33i01.3301922</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yu</surname> <given-names>B</given-names></string-name>, <string-name><surname>Yin</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>Spatio-temporal graph convolutional networks: a deep learning framework for traffic fore-casting</article-title>. <comment>arXiv:1709.04875. 2017</comment>. doi:<pub-id pub-id-type="doi">10.48550/arxiv.1709.04875</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ding</surname> <given-names>W</given-names></string-name>, <string-name><surname>Shi</surname> <given-names>Z</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Spatial-temporal complex graph convolution network for traffic flow prediction</article-title>. <source>Eng Appl Artif Intell</source>. <year>2023</year>;<volume>121</volume>(<issue>1</issue>):<fpage>106044</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.engappai.2023.106044</pub-id>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ma</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Lou</surname> <given-names>H</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>M</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>F</given-names></string-name>, <string-name><surname>Li</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Spatio-temporal fusion graph convolutional network for traffic flow forecasting</article-title>. <source>Inf Fusion</source>. <year>2024</year>;<volume>104</volume>:<fpage>102196</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.inffus.2023.102196</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Vaswani</surname> <given-names>A</given-names></string-name>, <string-name><surname>Shazeer</surname> <given-names>N</given-names></string-name>, <string-name><surname>Parmar</surname> <given-names>N</given-names></string-name>, <string-name><surname>Uszkoreit</surname> <given-names>J</given-names></string-name>, <string-name><surname>Jones</surname> <given-names>L</given-names></string-name>, <string-name><surname>Gomez</surname> <given-names>AN</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Attention is all you need</article-title>. In: <conf-name>Proceedings of the Neural Information Processing Systems 30 (NIPS 2017); 2017 Dec 4&#x2013;9</conf-name>; <publisher-loc>Long Beach, CA, USA</publisher-loc>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cai</surname> <given-names>L</given-names></string-name>, <string-name><surname>Janowicz</surname> <given-names>K</given-names></string-name>, <string-name><surname>Mai</surname> <given-names>G</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Traffic transformer: capturing the continuity and periodicity of time series for traffic forecasting</article-title>. <source>Trans GIS</source>. <year>2020</year>;<volume>24</volume>(<issue>3</issue>):<fpage>736</fpage>&#x2013;<lpage>55</lpage>. doi:<pub-id pub-id-type="doi">10.1111/tgis.12644</pub-id>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zheng</surname> <given-names>C</given-names></string-name>, <string-name><surname>Fan</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Qi</surname> <given-names>J</given-names></string-name></person-group>. <article-title>GMAN: a graph multi-attention network for traffic prediction</article-title>. <source>Proc AAAI Conf Artif Intell</source>. <year>2020</year>;<volume>34</volume>(<issue>1</issue>):<fpage>1234</fpage>&#x2013;<lpage>41</lpage>. doi:<pub-id pub-id-type="doi">10.1609/aaai.v34i01.5477</pub-id>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>C</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Bidirectional spatial-temporal adaptive transformer for urban traffic flow forecasting</article-title>. <source>IEEE Trans Neural Netw Learn Syst</source>. <year>2023</year>;<volume>34</volume>(<issue>10</issue>):<fpage>6913</fpage>&#x2013;<lpage>25</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TNNLS.2022.3183903</pub-id>; <pub-id pub-id-type="pmid">35771780</pub-id></mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Awan</surname> <given-names>FM</given-names></string-name>, <string-name><surname>Minerva</surname> <given-names>R</given-names></string-name>, <string-name><surname>Crespi</surname> <given-names>N</given-names></string-name></person-group>. <article-title>Using noise pollution data for traffic prediction in smart cities: experiments based on LSTM recurrent neural networks</article-title>. <source>IEEE Sens J</source>. <year>2021</year>;<volume>21</volume>(<issue>18</issue>):<fpage>20722</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JSEN.2021.3100324</pub-id>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>R</given-names></string-name>, <string-name><surname>Shahabi</surname> <given-names>C</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Diffusion convolutional recurrent neural network: data-driven traffic forecasting</article-title>. <comment>arXiv:1707.01926. 2017</comment>. doi:<pub-id pub-id-type="doi">10.48550/arxiv.1707.01926</pub-id>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhao</surname> <given-names>L</given-names></string-name>, <string-name><surname>Song</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>P</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>T</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>T-GCN: a temporal graph convolutional network for traffic prediction</article-title>. <source>IEEE Trans Intell Transp Syst</source>. <year>2019</year>;<volume>21</volume>(<issue>9</issue>):<fpage>3848</fpage>&#x2013;<lpage>58</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TITS.2019.2935152</pub-id>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>C</given-names></string-name></person-group>. <article-title>DSTGCN: dynamic spatial-temporal graph convolutional network for traffic prediction</article-title>. <source>IEEE Sens J</source>. <year>2022</year>;<volume>22</volume>(<issue>13</issue>):<fpage>13116</fpage>&#x2013;<lpage>24</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JSEN.2022.3176016</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Pan</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Cai</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhuang</surname> <given-names>B</given-names></string-name></person-group>. <article-title>Fast vision transformers with hilo attention</article-title>. In: <conf-name>Proceedings of the 36th Conference on Neural Information Processing Systems (NeurIPS 2022); 2022 Nov 28&#x2013;Dec 9</conf-name>; <publisher-loc>New Orleans, LA, USA</publisher-loc>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Park</surname> <given-names>N</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>S</given-names></string-name></person-group>. <article-title>How do vision transformers work?</article-title> <comment>arXiv:2202.06709. 2022</comment>. doi:<pub-id pub-id-type="doi">10.48550/arxiv.2202.06709</pub-id>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Feng</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>K</given-names></string-name></person-group>. <article-title>Low-high frequency network for spatial-temporal traffic flow forecasting</article-title>. <source>Eng Appl Artif Intell</source>. <year>2025</year>;<volume>158</volume>(<issue>11</issue>):<fpage>111304</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.engappai.2025.111304</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Jiang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Fan</surname> <given-names>J</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>X</given-names></string-name></person-group>. <article-title>Deep graph Gaussian processes for short-term traffic flow forecasting from spatiotemporal data</article-title>. <source>IEEE Trans Intell Transp Syst</source>. <year>2022</year>;<volume>23</volume>(<issue>11</issue>):<fpage>20177</fpage>&#x2013;<lpage>86</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TITS.2022.3178136</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Grohe</surname> <given-names>M</given-names></string-name></person-group>. <article-title>word2vec, node2vec, graph2vec, X2vec: towards a theory of vector embeddings of structured data</article-title>. In: <conf-name>Proceedings of the 39th ACM SIGMOD-SIGACT-SIGAI Symposium on Principles of Database Systems; 2020 Jun 14&#x2013;19</conf-name>; <publisher-loc>Portland, OR, USA</publisher-loc>. doi:<pub-id pub-id-type="doi">10.1145/3375395.3387641</pub-id>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wong</surname> <given-names>K</given-names></string-name>, <string-name><surname>Dornberger</surname> <given-names>R</given-names></string-name>, <string-name><surname>Hanne</surname> <given-names>T</given-names></string-name></person-group>. <article-title>An analysis of weight initialization methods in connection with different activation functions for feedforward neural networks</article-title>. <source>Evol Intell</source>. <year>2024</year>;<volume>17</volume>(<issue>3</issue>):<fpage>2081</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s12065-022-00795-y</pub-id>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Jiang</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zeng</surname> <given-names>P</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>K</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>W</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>H</given-names></string-name></person-group>. <article-title>FECAM: frequency enhanced channel attention mechanism for time series forecasting</article-title>. <source>Adv Eng Inform</source>. <year>2023</year>;<volume>58</volume>(<issue>8</issue>):<fpage>102158</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.aei.2023.102158</pub-id>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lin</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>T</given-names></string-name>, <string-name><surname>Cheng</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wen</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Image privacy protection scheme based on high-quality reconstruction DCT compression and nonlinear dynamics</article-title>. <source>Expert Syst Appl</source>. <year>2024</year>;<volume>257</volume>(<issue>5</issue>):<fpage>124891</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.eswa.2024.124891</pub-id>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Han</surname> <given-names>D</given-names></string-name>, <string-name><surname>Ye</surname> <given-names>T</given-names></string-name>, <string-name><surname>Han</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xia</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>P</given-names></string-name>, <etal>et al</etal></person-group>. <chapter-title>Agent attention: on the integration of softmax and linear attention</chapter-title>. In: <source>Computer Vision&#x2014;ECCV 2024</source>. <publisher-loc>Berlin/Heidelberg, Germany</publisher-loc>: <publisher-name>Springer</publisher-name>; <year>2024</year>. p. <fpage>124</fpage>&#x2013;<lpage>40</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-031-72973-7_8</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Song</surname> <given-names>C</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Spatial-temporal synchronous graph convolutional networks: a new framework for spatial-temporal network data forecasting</article-title>. <source>Proc AAAI Conf Artif Intell</source>. <year>2020</year>;<volume>34</volume>(<issue>1</issue>):<fpage>914</fpage>&#x2013;<lpage>21</lpage>. doi:<pub-id pub-id-type="doi">10.1609/aaai.v34i01.5438</pub-id>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Smith</surname> <given-names>BL</given-names></string-name>, <string-name><surname>Demetsky</surname> <given-names>MJ</given-names></string-name></person-group>. <article-title>Traffic flow forecasting: comparison of modeling approaches</article-title>. <source>J Transp Eng</source>. <year>1997</year>;<volume>123</volume>(<issue>4</issue>):<fpage>261</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1061/(asce)0733-947x(1997)123:4(261)</pub-id>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>Spatial-temporal fusion graph neural networks for traffic flow forecasting</article-title>. <source>Proc AAAI Conf Artif Intell</source>. <year>2021</year>;<volume>35</volume>(<issue>5</issue>):<fpage>4189</fpage>&#x2013;<lpage>96</lpage>. doi:<pub-id pub-id-type="doi">10.1609/aaai.v35i5.16542</pub-id>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Fu</surname> <given-names>K</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>X</given-names></string-name></person-group>. <chapter-title>Multi-view cascading spatial-temporal graph neural network for traffic flow forecasting</chapter-title>. In: <source>Artificial Neural Networks and Machine Learning&#x2014;ICANN 2022</source>. <publisher-loc>Berlin/Heidelberg, Germany</publisher-loc>: <publisher-name>Springer</publisher-name>; <year>2022</year>. p. <fpage>605</fpage>&#x2013;<lpage>16</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-031-15931-2_50</pub-id>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Guo</surname> <given-names>S</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>H</given-names></string-name>, <string-name><surname>Li</surname> <given-names>X</given-names></string-name>, <string-name><surname>Cong</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Learning dynamics and heterogeneity of spatial-temporal graph data for traffic forecasting</article-title>. <source>IEEE Trans Knowl Data Eng</source>. <year>2021</year>;<volume>34</volume>(<issue>11</issue>):<fpage>5415</fpage>&#x2013;<lpage>28</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TKDE.2021.3056502</pub-id>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Du</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>T</given-names></string-name>, <string-name><surname>Teng</surname> <given-names>F</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Li</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Multi-scale feature enhanced spatio-temporal learning for traffic flow forecasting</article-title>. <source>Knowl Based Syst</source>. <year>2024</year>;<volume>294</volume>(<issue>4</issue>):<fpage>111787</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.knosys.2024.111787</pub-id>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Leng</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Gated attention unit and mask attention network for traffic flow forecasting</article-title>. <source>Neural Comput Appl</source>. <year>2025</year>;<volume>37</volume>(<issue>20</issue>):<fpage>14889</fpage>&#x2013;<lpage>905</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s00521-025-11378-0</pub-id>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bai</surname> <given-names>D</given-names></string-name>, <string-name><surname>Xia</surname> <given-names>D</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Tian</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Future-heuristic differential graph transformer for traffic flow forecasting</article-title>. <source>Inf Sci</source>. <year>2025</year>;<volume>701</volume>(<issue>3</issue>):<fpage>121852</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.ins.2024.121852</pub-id>.</mixed-citation></ref>
</ref-list>
</back></article>