<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMES</journal-id>
<journal-id journal-id-type="nlm-ta">CMES</journal-id>
<journal-id journal-id-type="publisher-id">CMES</journal-id>
<journal-title-group>
<journal-title>Computer Modeling in Engineering &#x0026; Sciences</journal-title>
</journal-title-group>
<issn pub-type="epub">1526-1506</issn>
<issn pub-type="ppub">1526-1492</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">77726</article-id>
<article-id pub-id-type="doi">10.32604/cmes.2026.077726</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Gradient Descent with Time-Decaying Regularization for Training Linear Neural Networks</article-title>
<alt-title alt-title-type="left-running-head">Gradient Descent with Time-Decaying Regularization for Training Linear Neural Networks</alt-title>
<alt-title alt-title-type="right-running-head">Gradient Descent with Time-Decaying Regularization for Training Linear Neural Networks</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Palomino-Resendiz</surname><given-names>Sergio Isai</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-2" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Ulises Sol&#x00ED;s-Cervantes</surname><given-names>C&#x00E9;sar</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><email>csolisc@ipn.mx</email></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Cantera-Cantera</surname><given-names>Luis Alberto</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>de Jes&#x00FA;s Morales-Mercado</surname><given-names>Jorge</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Flores-Hern&#x00E1;ndez</surname><given-names>Diego Alonso</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<aff id="aff-1"><label>1</label><addr-line>Departamento de Ingenier&#x00ED;a en Control y Automatizaci&#x00F3;n</addr-line>, <institution>Escuela Superior de Ingenier&#x00ED;a Mec&#x00E1;nica y El&#x00E9;ctrica (ESIME), Unidad Zacatenco, Instituto Polit&#x00E9;cnico Nacional, Unidad Profesional Adolfo L&#x00F3;pez Mateos. Av. Luis Enrique Erro S/N</institution>, <addr-line>Gustavo A. Madero, Zacatenco, Ciudad de M&#x00E9;xico</addr-line>, <country>M&#x00E9;xico</country></aff>
<aff id="aff-2"><label>2</label><institution>Departamento de Control Autom&#x00E1;tico, Centro de Investigaci&#x00F3;n y de Estudios Avanzados (CINVESTAV) del Instituto Polit&#x00E9;cnico Nacional, Unidad Zacatenco, Av. Instituto Polit&#x00E9;cnico Nacional No. 2508</institution>, <addr-line>Col. San Pedro Zacatenco, Ciudad de M&#x00E9;xico</addr-line>, <country>M&#x00E9;xico</country></aff>
<aff id="aff-3"><label>3</label><institution>Facultad de Ingenier&#x00ED;a, Universidad An&#x00E1;huac M&#x00E9;xico, Campus Norte</institution>, <addr-line>Huixquilucan, Estado de M&#x00E9;xico</addr-line>, <country>M&#x00E9;xico</country></aff>
<aff id="aff-4"><label>4</label><institution>Secci&#x00F3;n de Estudios de Posgrado e Investigaci&#x00F3;n, Unidad Profesional Interdisciplinaria en Ingenier&#x00ED;a y Tecnolog&#x00ED;as Avanzadas (UPIITA), Instituto Polit&#x00E9;cnico Nacional, Av IPN 2580</institution>, <addr-line>La Laguna Ticoman, G. A. M</addr-line>., <addr-line>Ciudad de M&#x00E9;xico</addr-line>, <country>M&#x00E9;xico</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: C&#x00E9;sar Ulises Sol&#x00ED;s-Cervantes. Email: <email>csolisc@ipn.mx</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2026</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>27</day><month>4</month><year>2026</year>
</pub-date>
<volume>147</volume>
<issue>1</issue>
<elocation-id>26</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>12</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>25</day>
<month>02</month>
<year>2026</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2026 The Authors. Published by Tech Science Press.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>The Authors</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMES_77726.pdf"></self-uri>
<abstract>
<p>Many linear-in-parameters models arising in identification and control can be expressed as single-layer artificial neural networks (ANNs) with linear activation, enabling online learning via first-order optimization. In practice, however, standard gradient descent often exhibits slow convergence, large intermediate weights, and stagnation when the regressor data are ill-conditioned or computations are performed under finite precision. This paper proposes <italic>Gradient Descent with Time-Decaying Regularization</italic> (GD-TDR), a training algorithm that augments the quadratic loss with a regularization term whose weight decays exponentially in time. The proposed schedule enforces uniform strong convexity during early iterations, effectively mitigating neural-paralysis-like behavior associated with flat directions, while asymptotically vanishing so that the unregularized least-squares solution is recovered. A convergence theorem for GD-TDR is established and a concise pseudocode implementation is provided. Numerical and embedded experiments on an online identification problem of a Chua-type chaotic oscillator demonstrate that GD-TDR converges faster and avoids stagnation compared to standard gradient descent, without introducing the steady-state bias characteristic of fixed quadratic regularization.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Time-decaying regularization</kwd>
<kwd>gradient descent</kwd>
<kwd>single-layer linear neural network</kwd>
<kwd>online system identification</kwd>
<kwd>chaotic oscillator</kwd>
<kwd>embedded implementation</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>IPN-SIP</funding-source>
<award-id>SIP 20250023</award-id>
<award-id>20250424</award-id>
<award-id>20251300</award-id>
<award-id>20251721</award-id>
<award-id>20253411</award-id>
<award-id>MULTI-2026-0035</award-id>
</award-group>
<award-group id="awg2">
<funding-source>SECIHTI</funding-source>
<award-id>CF-2023-I-1635</award-id>
</award-group>
<award-group id="awg3">
<funding-source>istema Nacional de Investigadores e Investigadoras (SNII) of Mexico</funding-source>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<sec id="s1_1">
<label>1.1</label>
<title>State of the Art</title>
<p>Artificial neural networks (ANNs) have been extensively studied as flexible function approximators and parametric models across a wide range of scientific and engineering tasks. General surveys and application-driven reviews document the breadth of ANN deployments and motivate their use in identification and control, particularly when explicit first-principles modeling is difficult or when data-driven adaptation is required [<xref ref-type="bibr" rid="ref-1">1</xref>&#x2013;<xref ref-type="bibr" rid="ref-3">3</xref>]. Foundational treatments established the core modeling paradigms and training principles, including linear and nonlinear network structures and their algorithmic implementations [<xref ref-type="bibr" rid="ref-4">4</xref>&#x2013;<xref ref-type="bibr" rid="ref-6">6</xref>]. In this classical view, many practical learning rules can be interpreted as iterative optimization procedures acting on a quadratic or near-quadratic objective, a perspective that remains central to modern online learning formulations.</p>
<p>A large portion of the ANN training literature is rooted in incremental (first-order) updates. Recent theoretical analyses of learning dynamics in deep linear networks provide explicit characterizations of transient behavior and the interaction between initialization, regularization, and optimization geometry under gradient-based training [<xref ref-type="bibr" rid="ref-7">7</xref>,<xref ref-type="bibr" rid="ref-8">8</xref>]. Within the quadratic-loss setting, basic gradient-descent rules and their stochastic variants remain canonical examples of first-order adaptation mechanisms and highlight how data statistics, conditioning, and step-size choices shape stability and speed of learning [<xref ref-type="bibr" rid="ref-9">9</xref>,<xref ref-type="bibr" rid="ref-10">10</xref>]. In parallel, modern energy-based learning continues to emphasize the role of objective shaping and conditioning in learnability [<xref ref-type="bibr" rid="ref-11">11</xref>]. These works collectively support the view that optimization geometry&#x2014;not only model expressiveness&#x2014;plays a decisive role in whether training proceeds smoothly or becomes trapped in slow transient regimes.</p>
<p>Beyond standard multilayer architectures, several specialized ANN families have been developed for robustness, interpretability, and control-oriented deployment. Radial basis function networks and their robust variants provide a well-established pathway to stable approximation under uncertainty and noise [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-13">13</xref>]. In adaptive and self-learning control, ANN-based schemes have been reported for real-time compensation and online tuning, where training must remain stable under streaming data and limited numerical precision [<xref ref-type="bibr" rid="ref-14">14</xref>]. Related approaches in intelligent control also include fuzzy and cerebellar-model architectures that stress adaptation under nonlinearities and disturbances [<xref ref-type="bibr" rid="ref-15">15</xref>,<xref ref-type="bibr" rid="ref-16">16</xref>]. Complementary lines of research continue to refine computationally efficient first-order training, including recent advances in stochastic gradient descent variants [<xref ref-type="bibr" rid="ref-17">17</xref>].</p>
<p>A particularly demanding class of identification problems arises in nonlinear and chaotic dynamics, where sensitivity to initial conditions and measurement noise can degrade learning reliability. Recent studies illustrate both the feasibility of parameter identification in chaotic systems and the numerical challenges of learning from chaotic trajectories [<xref ref-type="bibr" rid="ref-18">18</xref>&#x2013;<xref ref-type="bibr" rid="ref-20">20</xref>]. These challenges are amplified in embedded or resource-constrained implementations, where finite-precision arithmetic and strict real-time requirements can exacerbate ill-conditioning and lead to slow or stagnant learning. Recent reviews on hardware realizations of neural methods, including FPGA-oriented implementations and embedded control applications, highlight the practical importance of training rules that remain stable and well-conditioned under limited precision. Consistent with these trends, widely used embedded platforms and rapid-prototyping toolchains have enabled end-to-end experimental validation of online learning strategies on microcontrollers [<xref ref-type="bibr" rid="ref-21">21</xref>,<xref ref-type="bibr" rid="ref-22">22</xref>].</p>
<p>Despite the breadth of architectures and applications, an enduring challenge in first-order online training is the susceptibility to slow plateaus and weight growth when the regressor is ill-conditioned or when optimization directions become nearly flat. Classical quadratic (L2/Tikhonov) regularization is a standard remedy to improve conditioning, yet fixed regularization may introduce steady-state bias when the target objective is the unregularized least-squares criterion. This motivates strategies that improve early-stage conditioning while preserving asymptotic fidelity to the original objective, which is the central perspective adopted in this work.</p>
</sec>
<sec id="s1_2">
<label>1.2</label>
<title>Description and Main Contributions</title>
<p>Single-layer ANNs with a linear activation function are equivalent to linear regression models and are widely used to represent linear-in-parameters structures in system identification, adaptive filtering, and control. In these applications, the model output can be written as a linear combination of known regressors and unknown parameters, so that training reduces to minimizing a least-squares functional. Although a closed-form solution exists for batch least squares, embedded and real-time settings often require iterative, lightweight, and online algorithms. For this reason, first-order methods based on gradient descent remain attractive due to their low computational complexity and ease of implementation [<xref ref-type="bibr" rid="ref-2">2</xref>,<xref ref-type="bibr" rid="ref-4">4</xref>,<xref ref-type="bibr" rid="ref-17">17</xref>].</p>
<p>In practice, however, standard gradient descent may perform poorly when the regressor data are ill-conditioned or nearly rank-deficient. These situations are frequent in online identification problems with delayed signals, correlated regressors, or limited excitation. The resulting cost surface can contain nearly flat directions, which leads to slow progress, long plateaus, and very large intermediate weights. On finite-precision hardware, such dynamics can manifest as training stagnation and numerical instabilities that are commonly described as neural-paralysis-like plateau behavior in the neural-network literature [<xref ref-type="bibr" rid="ref-23">23</xref>]. In the linear setting considered here, the effect is not caused by saturation of nonlinear activation functions, but rather by loss of curvature and poor conditioning of the quadratic objective.</p>
<p>A standard remedy is to augment the least-squares loss with a quadratic (Tikhonov) regularizer, which enforces strong convexity and penalizes large weights. Fixed regularization, however, introduces a bias: the minimizer of the regularized problem does not generally coincide with the minimizer of the original least-squares cost. This trade-off is particularly undesirable in identification tasks, where asymptotic accuracy is essential.</p>
<p>This work proposes GD-TDR (Gradient Descent Algorithm with Regularizer&#x2014;Time Decay), a first-order scheme that interpolates between these two extremes. The algorithm employs a quadratic regularizer whose coefficient decays exponentially over time. As a result, the early iterations benefit from improved curvature and bounded weights, while the regularization vanishes asymptotically and the algorithm recovers the minimizer of the unregularized least-squares functional.</p>
<p>The main contributions of this work are summarized as follows:<list list-type="bullet">
<list-item>
<p>a unified analytical framework that explicitly connects classical gradient descent, gradient descent with fixed quadratic (L2) regularization, and the proposed Gradient Descent with Time-Decaying Regularization (GD-TDR) through a single decay parameter, thereby clarifying their structural similarities and fundamental differences;</p></list-item>
<list-item>
<p>a rigorous convergence theorem establishing that the time-decaying regularization enforces uniform strong convexity during the transient phase while asymptotically recovering the minimizer of the unregularized least-squares problem;</p></list-item>
<list-item>
<p>a concise and self-contained pseudocode implementation of GD-TDR that directly reflects the theoretical development and facilitates reproducible implementation;</p></list-item>
<list-item>
<p>a comprehensive numerical and embedded validation, including online parameter identification of a Chua-type chaotic oscillator and a real-time implementation on an &#x00AE; STM32F4-Nucleo microcontroller, demonstrating accelerated convergence and mitigation of stagnation without steady-state bias.</p></list-item>
</list></p>
<p>The paper is organized as follows. <xref ref-type="sec" rid="s2">Section 2</xref> introduces the linear ANN model and the least-squares training objective. <xref ref-type="sec" rid="s3">Section 3</xref> compares gradient-descent training schemes and presents GD-TDR. <xref ref-type="sec" rid="s4">Section 4</xref> states and proves the convergence theorem and provides the GD-TDR pseudocode. <xref ref-type="sec" rid="s5">Section 5</xref> shows how common identification models can be written in linear ANN form. <xref ref-type="sec" rid="s6">Sections 6</xref> and <xref ref-type="sec" rid="s7">7</xref> report the numerical and embedded validation, respectively. <xref ref-type="sec" rid="s8">Section 8</xref> concludes the paper and outlines future work.</p>
</sec>
</sec>
<sec id="s2">
<label>2</label>
<title>Single-Layer Linear ANN and Least-Squares Training</title>
<sec id="s2_1">
<label>2.1</label>
<title>Model Representation</title>
<p>Consider a single-layer ANN with linear activation (<italic>purelin</italic>) and no bias. Its output is
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mo>:=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow></mml:mrow><mml:mo>,</mml:mo></mml:math></disp-formula>where the input vector is <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow></mml:mrow><mml:mo>:=</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x22EF;</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>]</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula> and the weight vector is <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo>:=</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x22EF;</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>]</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula>. <xref ref-type="fig" rid="fig-1">Fig. 1</xref> illustrates the architecture.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Single-layer ANN with linear activation.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-1.tif"/>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Batch Least-Squares Objective</title>
<p>Given a finite data set <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:msubsup><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup></mml:math></inline-formula>, define the prediction errors <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:msub><mml:mi>e</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>:=</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and the quadratic cost
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>:=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:msubsup><mml:mi>e</mml:mi><mml:mi>k</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>Let <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi>&#x02133;</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> be the regressor matrix and <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow></mml:mrow><mml:mi>N</mml:mi></mml:msup></mml:math></inline-formula> the target vector,
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mo>:=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow></mml:mrow><mml:mn>1</mml:mn></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x22EF;</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow></mml:mrow><mml:mi>N</mml:mi></mml:msub><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>:=</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x22EF;</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mo>]</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>Then
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo symmetric="true" maxsize="1.2em" minsize="1.2em">&#x2016;</mml:mo></mml:mrow></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:msubsup><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo symmetric="true" maxsize="1.2em" minsize="1.2em">&#x2016;</mml:mo></mml:mrow></mml:mstyle><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>The gradient and Hessian of <italic>J</italic> with respect to <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:math></inline-formula> are
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mspace width="thinmathspace" /><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mstyle><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:msup><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mspace width="thinmathspace" /><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>The Hessian is positive semidefinite. It is positive definite (and hence <italic>J</italic> is strongly convex) if and only if <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">XX</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula> is nonsingular.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Training Schemes and Algorithmic Comparisons</title>
<p>This section summarizes three closely related first-order schemes: classical gradient descent (GD), gradient descent with a <italic>fixed</italic> quadratic regularizer (GD-QR), and the proposed time-decaying regularized scheme (GD-TDR). Only GD-TDR is presented in pseudocode form (<xref ref-type="sec" rid="s4">Section 4</xref>).</p>
<sec id="s3_1">
<label>3.1</label>
<title>Classical Gradient Descent (GD)</title>
<p>Using <xref ref-type="disp-formula" rid="eqn-2">(2)</xref>, the GD update with step size <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:mi>&#x03B7;</mml:mi><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> is
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mspace width="thinmathspace" /><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mi>J</mml:mi><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow></mml:mstyle><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mstyle><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mi>&#x03B7;</mml:mi></mml:mrow><mml:mi>N</mml:mi></mml:mfrac><mml:mspace width="thinmathspace" /><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mspace width="thinmathspace" /><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">e</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">e</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>:=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>. When <italic>J</italic> is strongly convex and <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:mi>&#x03B7;</mml:mi></mml:math></inline-formula> is chosen appropriately, GD converges linearly to the unique minimizer. If <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">XX</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula> is singular or ill-conditioned, <italic>J</italic> is not strongly convex and GD may exhibit slow progress and large intermediate weights.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Gradient Descent with Fixed Quadratic Regularization (GD-QR)</title>
<p>A standard approach to improve conditioning is to add a quadratic regularizer
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mi>&#x03B3;</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>:=</mml:mo><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mfrac><mml:mi>&#x03B3;</mml:mi><mml:mn>2</mml:mn></mml:mfrac><mml:mspace width="thinmathspace" /><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mi>&#x03B3;</mml:mi><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> and <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi>&#x02133;</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is symmetric positive definite. The Hessian becomes
<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:msup><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mi>&#x03B3;</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mspace width="thinmathspace" /><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:mrow><mml:mo>&#x227B;</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo></mml:math></disp-formula>so <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mi>&#x03B3;</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is strongly convex for any <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mi>&#x03B3;</mml:mi><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula>. The GD-QR update reads
<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mspace width="thinmathspace" /><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">e</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>Fixed regularization controls weight growth and improves curvature, but the minimizer of <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mi>&#x03B3;</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> generally differs from the minimizer of <italic>J</italic>. In identification problems, this bias can be detrimental.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Proposed GD-TDR (Time-Decaying Quadratic Regularization)</title>
<p>GD-TDR replaces the constant regularization weight <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mi>&#x03B3;</mml:mi></mml:math></inline-formula> in <xref ref-type="disp-formula" rid="eqn-9">(9)</xref> by a time-decaying sequence
<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>:=</mml:mo><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:msup><mml:mi>&#x03BB;</mml:mi><mml:mi>j</mml:mi></mml:msup><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:mn>0</mml:mn><mml:mo>&#x003C;</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mo>&#x003C;</mml:mo><mml:mn>1.</mml:mn></mml:math></disp-formula></p>
<p>The weight update becomes
<disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mspace width="thinmathspace" /><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">e</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>Two limiting cases highlight the relationship among the three schemes:<list list-type="bullet">
<list-item>
<p><italic>Classical GD:</italic> setting <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> yields <xref ref-type="disp-formula" rid="eqn-6">(6)</xref>.</p></list-item>
<list-item>
<p><italic>Fixed regularization:</italic> setting <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:mi>&#x03BB;</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> yields GD-QR in <xref ref-type="disp-formula" rid="eqn-9">(9)</xref>.</p></list-item>
</list></p>
<p>Therefore, GD-TDR provides a continuous mechanism to improve conditioning early in training while asymptotically removing the regularization bias. The theoretical properties of this scheme are stated in Theorem 1.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Theoretical Properties and GD-TDR Pseudocode</title>
<p>We restate the objective in compact form. Define
<disp-formula id="eqn-12"><label>(12)</label><mml:math id="mml-eqn-12" display="block"><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>:=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo symmetric="true" maxsize="1.2em" minsize="1.2em">&#x2016;</mml:mo></mml:mrow></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:msubsup><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo symmetric="true" maxsize="1.2em" minsize="1.2em">&#x2016;</mml:mo></mml:mrow></mml:mstyle><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">e</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>:=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:math></disp-formula>and let <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:mrow><mml:mo>&#x227B;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> be symmetric. For each iteration <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:mi>j</mml:mi></mml:math></inline-formula> consider the regularized functional
<disp-formula id="eqn-13"><label>(13)</label><mml:math id="mml-eqn-13" display="block"><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>:=</mml:mo><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mfrac><mml:mrow><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mfrac><mml:mspace width="thinmathspace" /><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo>,</mml:mo></mml:math></disp-formula>with <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> defined by <xref ref-type="disp-formula" rid="eqn-10">(10)</xref>.</p>
<p><bold>Theorem 1 (Online GD-TDR):</bold> <italic>Let <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">M</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> and <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:msub><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow></mml:mrow><mml:mi>N</mml:mi></mml:msup></mml:math></inline-formula> be given and assume that <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">XX</mml:mtext></mml:mrow></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup></mml:math></inline-formula> is positive semidefinite. Let <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">M</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> be symmetric and positive definite and define</italic>
<disp-formula id="eqn-14"><label>(14)</label><mml:math id="mml-eqn-14" display="block"><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow></mml:mrow><mml:mo>:=</mml:mo><mml:mfrac><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mspace width="thinmathspace" /><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p><italic>For <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> and <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mn>0</mml:mn><mml:mo>&#x003C;</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mo>&#x003C;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> define</italic>
<disp-formula id="eqn-15"><label>(15)</label><mml:math id="mml-eqn-15" display="block"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>:=</mml:mo><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:msup><mml:mi>&#x03BB;</mml:mi><mml:mi>j</mml:mi></mml:msup><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo></mml:math></disp-formula><italic>and</italic>
<disp-formula id="eqn-16"><label>(16)</label><mml:math id="mml-eqn-16" display="block"><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>:=</mml:mo><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mfrac><mml:mrow><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mfrac><mml:mspace width="thinmathspace" /><mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>(a) <italic>For every <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:mi>j</mml:mi><mml:mo>&#x2265;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> the functional <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula> is strongly convex. More precisely, its Hessian</italic><disp-formula id="eqn-17"><label>(17)</label><mml:math id="mml-eqn-17" display="block"><mml:msubsup><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mrow><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:mrow></mml:math></disp-formula><italic>satisfies</italic>
<disp-formula id="eqn-18"><label>(18)</label><mml:math id="mml-eqn-18" display="block"><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mspace width="negativethinmathspace" /><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mrow><mml:mspace width="negativethinmathspace" /><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2265;</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mspace width="thinmathspace" /><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></disp-formula><italic>for all <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mi>n</mml:mi></mml:msup></mml:math></inline-formula>. Consequently, each <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula> has a unique global minimizer <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:msubsup><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>j</mml:mi><mml:mrow><mml:mo>&#x22C6;</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula></italic>.</p>
<p>(b) <italic>Suppose that <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:msup><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup></mml:math></inline-formula> is positive definite, so that J has a unique minimizer <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:msup><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mrow><mml:mo>&#x22C6;</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula>. Let <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:mi>&#x03BC;</mml:mi><mml:mo>:=</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> and <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:mi>L</mml:mi><mml:mo>:=</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, and denote by <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> and <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> the extremal eigenvalues of <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:math></inline-formula>. Choose <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:mi>&#x03B7;</mml:mi><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> such that</italic>
<disp-formula id="eqn-19"><label>(19)</label><mml:math id="mml-eqn-19" display="block"><mml:mn>0</mml:mn><mml:mo>&#x003C;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mo>&#x003C;</mml:mo><mml:mfrac><mml:mn>2</mml:mn><mml:mrow><mml:mi>L</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p><italic>Consider the GD-TDR iteration</italic>
<disp-formula id="eqn-20"><label>(20)</label><mml:math id="mml-eqn-20" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mtext mathvariant="bold">e</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-21"><label>(21)</label><mml:math id="mml-eqn-21" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mspace width="thinmathspace" /><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mrow><mml:mtext mathvariant="bold">e</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula><italic>Then <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x2265;</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is bounded and converges to <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:msup><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mrow><mml:mo>&#x22C6;</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula></italic>.</p>
<p>(c) <italic>For each <inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:mi>j</mml:mi></mml:math></inline-formula> and every <inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mi>n</mml:mi></mml:msup></mml:math></inline-formula></italic>,
<disp-formula id="eqn-22"><label>(22)</label><mml:math id="mml-eqn-22" display="block"><mml:msub><mml:mrow><mml:mo symmetric="true">&#x2016;</mml:mo><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo symmetric="true">&#x2016;</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msub><mml:mo>&#x2265;</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mspace width="thinmathspace" /><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mspace width="thinmathspace" /><mml:msub><mml:mrow><mml:mo symmetric="true">&#x2016;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msubsup><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>j</mml:mi><mml:mrow><mml:mo>&#x22C6;</mml:mo></mml:mrow></mml:msubsup><mml:mo symmetric="true">&#x2016;</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msub><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p><italic>In particular, when <inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is large (early iterations), the curvature of <inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula> is uniformly bounded away from zero and the gradient cannot vanish far from the minimizer. This reduces extended flat regions of the cost surface and mitigates neural-paralysis-like stagnation associated with weight growth</italic>.</p>
<p><bold>Proof:</bold> (a) Since <inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow></mml:math></inline-formula> is symmetric and positive semidefinite and <inline-formula id="ieqn-50"><mml:math id="mml-ieqn-50"><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo>&#x227B;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula>, the matrix <inline-formula id="ieqn-51"><mml:math id="mml-ieqn-51"><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:math></inline-formula> is symmetric positive definite for every <inline-formula id="ieqn-52"><mml:math id="mml-ieqn-52"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula>. Moreover,
<disp-formula id="eqn-23"><label>(23)</label><mml:math id="mml-eqn-23" display="block"><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2265;</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2265;</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x003E;</mml:mo><mml:mn>0.</mml:mn></mml:math></disp-formula></p>
<p>Hence <inline-formula id="ieqn-53"><mml:math id="mml-ieqn-53"><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula> is strongly convex and has a unique minimizer <inline-formula id="ieqn-54"><mml:math id="mml-ieqn-54"><mml:msubsup><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>j</mml:mi><mml:mrow><mml:mo>&#x22C6;</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula>.</p>
<p>(b) When <inline-formula id="ieqn-55"><mml:math id="mml-ieqn-55"><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow></mml:math></inline-formula> is positive definite, <italic>J</italic> is <inline-formula id="ieqn-56"><mml:math id="mml-ieqn-56"><mml:mi>&#x03BC;</mml:mi></mml:math></inline-formula>-strongly convex and <italic>L</italic>-smooth. For each <inline-formula id="ieqn-57"><mml:math id="mml-ieqn-57"><mml:mi>j</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-58"><mml:math id="mml-ieqn-58"><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula> is <inline-formula id="ieqn-59"><mml:math id="mml-ieqn-59"><mml:msub><mml:mi>&#x03BC;</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula>-strongly convex and <inline-formula id="ieqn-60"><mml:math id="mml-ieqn-60"><mml:msub><mml:mi>L</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula>-smooth with
<disp-formula id="eqn-24"><label>(24)</label><mml:math id="mml-eqn-24" display="block"><mml:msub><mml:mi>&#x03BC;</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2265;</mml:mo><mml:mi>&#x03BC;</mml:mi><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:msub><mml:mi>L</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">H</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2264;</mml:mo><mml:mi>L</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>Thus <inline-formula id="ieqn-61"><mml:math id="mml-ieqn-61"><mml:mn>0</mml:mn><mml:mo>&#x003C;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mo>&#x003C;</mml:mo><mml:mn>2</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula> holds uniformly for all <inline-formula id="ieqn-62"><mml:math id="mml-ieqn-62"><mml:mi>j</mml:mi></mml:math></inline-formula>. The update can be written as
<disp-formula id="eqn-25"><label>(25)</label><mml:math id="mml-eqn-25" display="block"><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mspace width="thinmathspace" /><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow></mml:mstyle><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mstyle><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>Gradient descent on a strongly convex, smooth function with a step size in <inline-formula id="ieqn-63"><mml:math id="mml-ieqn-63"><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is a contraction mapping, hence <inline-formula id="ieqn-64"><mml:math id="mml-ieqn-64"><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula> is bounded. Since <inline-formula id="ieqn-65"><mml:math id="mml-ieqn-65"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">&#x2192;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula>, <inline-formula id="ieqn-66"><mml:math id="mml-ieqn-66"><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">&#x2192;</mml:mo><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> uniformly on bounded sets. Taking limits in the iteration yields <inline-formula id="ieqn-67"><mml:math id="mml-ieqn-67"><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mover><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> for the limit point <inline-formula id="ieqn-68"><mml:math id="mml-ieqn-68"><mml:mrow><mml:mover><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow></mml:math></inline-formula>, which must equal the unique minimizer <inline-formula id="ieqn-69"><mml:math id="mml-ieqn-69"><mml:msup><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mrow><mml:mo>&#x22C6;</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula>.</p>
<p>(c) By strong convexity of <inline-formula id="ieqn-70"><mml:math id="mml-ieqn-70"><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula> with modulus at least <inline-formula id="ieqn-71"><mml:math id="mml-ieqn-71"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> and the standard gradient characterization of strong convexity,
<disp-formula id="eqn-26"><label>(26)</label><mml:math id="mml-eqn-26" display="block"><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo symmetric="true" maxsize="1.2em" minsize="1.2em">&#x2016;</mml:mo></mml:mrow></mml:mstyle><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo symmetric="true" maxsize="1.2em" minsize="1.2em">&#x2016;</mml:mo></mml:mrow></mml:mstyle><mml:mn>2</mml:mn></mml:msub><mml:mo>&#x2265;</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mspace width="thinmathspace" /><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo symmetric="true" maxsize="1.2em" minsize="1.2em">&#x2016;</mml:mo></mml:mrow></mml:mstyle><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msubsup><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>j</mml:mi><mml:mrow><mml:mo>&#x22C6;</mml:mo></mml:mrow></mml:msubsup><mml:msub><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo symmetric="true" maxsize="1.2em" minsize="1.2em">&#x2016;</mml:mo></mml:mrow></mml:mstyle><mml:mn>2</mml:mn></mml:msub><mml:mo>.</mml:mo></mml:math></disp-formula> <inline-formula id="ieqn-72"><mml:math id="mml-ieqn-72"><mml:mi>&#x25FB;</mml:mi></mml:math></inline-formula></p>
<p><bold>Remark 1:</bold> <italic>The decay rule <inline-formula id="ieqn-73"><mml:math id="mml-ieqn-73"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> implements an exponentially vanishing regularization weight. Early in training, <inline-formula id="ieqn-74"><mml:math id="mml-ieqn-74"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> can be chosen sufficiently large so that the quadratic term dominates the curvature and penalizes large weights. As <inline-formula id="ieqn-75"><mml:math id="mml-ieqn-75"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> decreases, the regularization bias disappears asymptotically and the algorithm recovers the geometry of the original least-squares cost. An alternative and simplified way to visualize all of the above is through the pseudocode of the algorithm contained in Algorithm 1</italic>.</p>
<fig id="fig-13">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-13.tif"/>
</fig>
<p>To make the above easier to visualize, <xref ref-type="table" rid="table-1">Table 1</xref> presents a comparison of key aspects.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Comparison of GD, GD&#x2013;QR, and GD&#x2013;TDR for quadratic objectives.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>Aspect</th>
<th>GD</th>
<th>GD&#x2013;QR (fixed)</th>
<th>GD&#x2013;TDR (time-decaying)</th>
</tr>
</thead>
<tbody>
<tr>
<td>Objective optimized</td>
<td><inline-formula id="ieqn-94"><mml:math id="mml-ieqn-94"><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac></mml:mstyle><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:msup><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow><mml:msubsup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:math></inline-formula></td>
<td><inline-formula id="ieqn-95"><mml:math id="mml-ieqn-95"><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mi>&#x03BB;</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:msubsup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:math></inline-formula></td>
<td><inline-formula id="ieqn-96"><mml:math id="mml-ieqn-96"><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mi>&#x03BB;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:msubsup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:math></inline-formula></td>
</tr>
<tr>
<td>Update rule</td>
<td><inline-formula id="ieqn-97"><mml:math id="mml-ieqn-97"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula></td>
<td><inline-formula id="ieqn-98"><mml:math id="mml-ieqn-98"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow></mml:mstyle><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x03BB;</mml:mi><mml:mspace width="thinmathspace" /><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula></td>
<td><inline-formula id="ieqn-99"><mml:math id="mml-ieqn-99"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow></mml:mstyle><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mi>J</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x03BB;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mspace width="thinmathspace" /><mml:msub><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula></td>
</tr>
<tr>
<td>Early strong convexity</td>
<td>Not guaranteed (data dependent)</td>
<td>Guaranteed for <inline-formula id="ieqn-100"><mml:math id="mml-ieqn-100"><mml:mi>&#x03BB;</mml:mi><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula></td>
<td>Guaranteed if <inline-formula id="ieqn-101"><mml:math id="mml-ieqn-101"><mml:mi>&#x03BB;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is sufficiently large</td>
</tr>
<tr>
<td>Hessian conditioning</td>
<td><inline-formula id="ieqn-102"><mml:math id="mml-ieqn-102"><mml:msup><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mi>J</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="false" scriptlevel="0"><mml:mfrac><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:mfrac></mml:mstyle><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:msup><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> (may be ill-conditioned)</td>
<td><inline-formula id="ieqn-103"><mml:math id="mml-ieqn-103"><mml:msup><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mi>&#x03BB;</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="false" scriptlevel="0"><mml:mfrac><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:mfrac></mml:mstyle><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:msup><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mtext mathvariant="bold">I</mml:mtext></mml:mrow></mml:math></inline-formula></td>
<td><inline-formula id="ieqn-104"><mml:math id="mml-ieqn-104"><mml:msup><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mi>&#x03BB;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="false" scriptlevel="0"><mml:mfrac><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:mfrac></mml:mstyle><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:msup><mml:mrow><mml:mtext mathvariant="bold">X</mml:mtext></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x03BB;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mtext mathvariant="bold">I</mml:mtext></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td>Asymptotic objective recovered</td>
<td>Yes (minimizes <italic>J</italic>)</td>
<td>No (minimizes <inline-formula id="ieqn-105"><mml:math id="mml-ieqn-105"><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mi>&#x03BB;</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>)</td>
<td>Yes, if <inline-formula id="ieqn-106"><mml:math id="mml-ieqn-106"><mml:mi>&#x03BB;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">&#x2192;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula></td>
</tr>
<tr>
<td>Steady-state bias (w.r.t. LS)</td>
<td>No</td>
<td>Yes (increases with <inline-formula id="ieqn-107"><mml:math id="mml-ieqn-107"><mml:mi>&#x03BB;</mml:mi></mml:math></inline-formula>)</td>
<td>No (vanishing regularization)</td>
</tr>
<tr>
<td>Stagnation/NP mitigation</td>
<td>May exhibit plateaus</td>
<td>Plateaus reduced, but biased optimum</td>
<td>Plateaus reduced without biasing the optimum</td>
</tr>
<tr>
<td>Recommended use case</td>
<td>Well-conditioned regressors</td>
<td>Permanent conditioning / shrinkage</td>
<td>Early conditioning with unbiased asymptote</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5">
<label>5</label>
<title>Application to Online Parameter Identification</title>
<p>This section shows how common linear identification models can be written as single-layer linear ANNs, which allows applying GD-TDR directly.</p>
<sec id="s5_1">
<label>5.1</label>
<title>Discrete Transfer Functions</title>
<p>Consider a discrete transfer function
<disp-formula id="eqn-27"><label>(27)</label><mml:math id="mml-eqn-27" display="block"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:mi>Y</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>z</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>z</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:mstyle><mml:mo>:=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mi>z</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msup><mml:mi>z</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mo>&#x22EF;</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msup><mml:mi>z</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mi>z</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msup><mml:mi>z</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>m</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mo>&#x22EF;</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msup><mml:mi>z</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac></mml:mstyle><mml:mo>,</mml:mo></mml:math></disp-formula>with <inline-formula id="ieqn-108"><mml:math id="mml-ieqn-108"><mml:mi>s</mml:mi><mml:mo>&#x2264;</mml:mo><mml:mi>m</mml:mi></mml:math></inline-formula>. After inverse <inline-formula id="ieqn-109"><mml:math id="mml-ieqn-109"><mml:mi>z</mml:mi></mml:math></inline-formula>-transform (zero initial conditions), one obtains an ARX-like representation
<disp-formula id="eqn-28"><label>(28)</label><mml:math id="mml-eqn-28" display="block"><mml:mi>y</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>=</mml:mo><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mi>a</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mspace width="thinmathspace" /><mml:mi>u</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>&#x2212;</mml:mo><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mi>b</mml:mi><mml:mi>r</mml:mi></mml:msub><mml:mspace width="thinmathspace" /><mml:mi>y</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>Defining
<disp-formula id="eqn-29"><label>(29)</label><mml:math id="mml-eqn-29" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>:=</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:mtable columnalign="center center center center center center center" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:msub><mml:mi>a</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>a</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mtd><mml:mtd><mml:mo>&#x22EF;</mml:mo></mml:mtd><mml:mtd><mml:msub><mml:mi>a</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mtd><mml:mtd><mml:mo>&#x22EF;</mml:mo></mml:mtd><mml:mtd><mml:msub><mml:mi>b</mml:mi><mml:mi>m</mml:mi></mml:msub></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>:=</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:mtable columnalign="center center center center center center center" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mi>u</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:mtd><mml:mtd><mml:mi>u</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x22EF;</mml:mo></mml:mtd><mml:mtd><mml:mi>u</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x22EF;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>The model becomes <inline-formula id="ieqn-110"><mml:math id="mml-ieqn-110"><mml:mi>y</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mrow><mml:mo>&#x22BA;</mml:mo></mml:mrow></mml:msup><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula>, which is exactly the output of a single-layer linear ANN. <xref ref-type="fig" rid="fig-2">Fig. 2</xref> sketches this identification setup.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Single-layer ANN representation for parameter identification.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-2.tif"/>
</fig>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Discrete State-Space Models</title>
<p>For a discrete state-space (DSS) system
<disp-formula id="eqn-30"><label>(30)</label><mml:math id="mml-eqn-30" display="block"><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mtext mathvariant="bold">A</mml:mtext></mml:mrow><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>+</mml:mo><mml:mrow><mml:mtext mathvariant="bold">B</mml:mtext></mml:mrow><mml:mrow><mml:mtext mathvariant="bold">u</mml:mtext></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula>with appropriate dimensions, the right-hand side is linear in the unknown entries of <inline-formula id="ieqn-111"><mml:math id="mml-ieqn-111"><mml:mrow><mml:mtext mathvariant="bold">A</mml:mtext></mml:mrow></mml:math></inline-formula> and <inline-formula id="ieqn-112"><mml:math id="mml-ieqn-112"><mml:mrow><mml:mtext mathvariant="bold">B</mml:mtext></mml:mrow></mml:math></inline-formula>. By stacking the parameters into a single vector (or matrix) and defining a regressor vector that contains state and input components, the DSS update can also be written in the form <inline-formula id="ieqn-113"><mml:math id="mml-ieqn-113"><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mspace width="thinmathspace" /><mml:mi mathvariant="bold-italic">&#x03D5;</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula>, where <inline-formula id="ieqn-114"><mml:math id="mml-ieqn-114"><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow></mml:math></inline-formula> collects the unknown parameters. This is compatible with GD-TDR, which can be applied entrywise.</p>
<p>A practical issue is causality: to update parameters at time <inline-formula id="ieqn-115"><mml:math id="mml-ieqn-115"><mml:mi>n</mml:mi></mml:math></inline-formula>, the target <inline-formula id="ieqn-116"><mml:math id="mml-ieqn-116"><mml:mrow><mml:mtext mathvariant="bold">x</mml:mtext></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> is needed. A simple remedy is to insert a one-step delay in the learning loop, which preserves the identification objective while keeping the update implementable in real time. <xref ref-type="fig" rid="fig-3">Fig. 3</xref> illustrates the ANN view of the DSS model, while <xref ref-type="fig" rid="fig-4">Fig. 4</xref> shows a delay-based identification for DSS parameter identification.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>DSS model viewed as a linear ANN mapping.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-3.tif"/>
</fig><fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Delay-based scheme for DSS parameter identification.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-4.tif"/>
</fig>

<p><bold>Remark 2:</bold> <italic>Although the focus is on linear-in-parameters models, the same idea can be used to identify local linearizations of nonlinear systems around operating points, provided the regressors are constructed accordingly</italic>.</p>

</sec>
</sec>
<sec id="s6">
<label>6</label>
<title>Numerical Validation</title>
<sec id="s6_1">
<label>6.1</label>
<title>Experimental Setup</title>
<p>The numerical validation considers online parameter identification of a chaotic system whose dynamics are equivalent to those of a Chua-type oscillator. Chaotic trajectories provide a demanding excitation pattern for adaptive algorithms and are known to expose slow transients and stagnation effects in gradient-based learning [<xref ref-type="bibr" rid="ref-19">19</xref>,<xref ref-type="bibr" rid="ref-20">20</xref>].</p>
<p>The Chua oscillator is described by
<disp-formula id="eqn-31"><label>(31)</label><mml:math id="mml-eqn-31" display="block"><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left left" rowspacing=".2em" columnspacing="1em" displaystyle="false"><mml:mtr><mml:mtd><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo>&#x02D9;</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x03B1;</mml:mi><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow></mml:mstyle><mml:mi>y</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>x</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03C6;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mstyle><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo>&#x02D9;</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mi>x</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>y</mml:mi><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mover><mml:mi>z</mml:mi><mml:mo>&#x02D9;</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mi>y</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:math></disp-formula>where
<disp-formula id="eqn-32"><label>(32)</label><mml:math id="mml-eqn-32" display="block"><mml:mi>&#x03C6;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>:=</mml:mo><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow></mml:mstyle><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mstyle><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>For the parameter values <inline-formula id="ieqn-117"><mml:math id="mml-ieqn-117"><mml:mi>&#x03B1;</mml:mi><mml:mo>=</mml:mo><mml:mn>15.6</mml:mn></mml:math></inline-formula>, <inline-formula id="ieqn-118"><mml:math id="mml-ieqn-118"><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mn>25</mml:mn></mml:math></inline-formula>, <inline-formula id="ieqn-119"><mml:math id="mml-ieqn-119"><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mn>1.1429</mml:mn></mml:math></inline-formula>, and <inline-formula id="ieqn-120"><mml:math id="mml-ieqn-120"><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mn>0.7143</mml:mn></mml:math></inline-formula>, the corresponding attractor is shown in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Trajectory of Chua dynamics (chaotic attractor).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-5.tif"/>
</fig>
<p>To pose an identification problem that is linear in the unknown parameters, the dynamics are rewritten as
<disp-formula id="eqn-33"><label>(33)</label><mml:math id="mml-eqn-33" display="block"><mml:munder><mml:mrow><mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo>&#x02D9;</mml:mo></mml:mover></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo>&#x02D9;</mml:mo></mml:mover></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mover><mml:mi>z</mml:mi><mml:mo>&#x02D9;</mml:mo></mml:mover></mml:mrow></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x23DF;</mml:mo></mml:munder></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mtext mathvariant="bold">y</mml:mtext></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:munder><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mtable columnalign="center center center center" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>11</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>12</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>13</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>14</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>21</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>22</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>23</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>24</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>31</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>32</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>33</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mn>34</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x23DF;</mml:mo></mml:munder></mml:mrow><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow></mml:mrow></mml:munder><mml:munder><mml:mrow><mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mi>x</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>y</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>z</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>d</mml:mi></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x23DF;</mml:mo></mml:munder></mml:mrow><mml:mrow><mml:mi mathvariant="bold-italic">&#x03D5;</mml:mi></mml:mrow></mml:munder><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:mi>d</mml:mi><mml:mo>:=</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:math></disp-formula></p>
<p>In this form, <inline-formula id="ieqn-121"><mml:math id="mml-ieqn-121"><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow></mml:math></inline-formula> is an unknown parameter matrix to be estimated online from the measured signals. The validation compares GD-TDR against classical GD. In addition, <xref ref-type="sec" rid="s3">Section 3</xref> provides an explicit comparison with fixed quadratic regularization (GD-QR), which is obtained as the special case <inline-formula id="ieqn-122"><mml:math id="mml-ieqn-122"><mml:mi>&#x03BB;</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>.</p>
<p><xref ref-type="fig" rid="fig-6">Fig. 6</xref> shows the main program used in the numerical simulations.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Main program in &#x00AE; Matlab-Simulink environment.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-6.tif"/>
</fig>
<p>In the reported tests, the learning factor was set to <inline-formula id="ieqn-123"><mml:math id="mml-ieqn-123"><mml:mi>&#x03B7;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.02</mml:mn></mml:math></inline-formula>. For GD-TDR, the regularization parameters were initialized at <inline-formula id="ieqn-124"><mml:math id="mml-ieqn-124"><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:mn>0.99</mml:mn></mml:math></inline-formula> and decayed according to <xref ref-type="disp-formula" rid="eqn-10">(10)</xref> with <inline-formula id="ieqn-125"><mml:math id="mml-ieqn-125"><mml:mi>&#x03BB;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.9998</mml:mn></mml:math></inline-formula>. The simulations used the <monospace>ode8</monospace> solver with fixed step size and sampling time <inline-formula id="ieqn-126"><mml:math id="mml-ieqn-126"><mml:mn>0.001</mml:mn><mml:mspace width="thinmathspace" /><mml:mtext>s</mml:mtext></mml:math></inline-formula>.</p>
</sec>
<sec id="s6_2">
<label>6.2</label>
<title>Results</title>
<p><xref ref-type="fig" rid="fig-7">Fig. 7</xref> shows the convergence of the estimated parameters towards the target values, while <xref ref-type="fig" rid="fig-8">Fig. 8</xref> reports the norm of the estimation error. The final identified parameters for GD-TDR, GD and GD-QR are listed in <xref ref-type="disp-formula" rid="eqn-34">(34)</xref> to <xref ref-type="disp-formula" rid="eqn-36">(36)</xref>, respectively.</p>

<p><disp-formula id="eqn-34"><label>(34)</label><mml:math id="mml-eqn-34" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>GD-TDR</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mtable columnalign="left left left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>4.457</mml:mn></mml:mtd><mml:mtd><mml:mn>15.600</mml:mn></mml:mtd><mml:mtd><mml:mn>0.000</mml:mn></mml:mtd><mml:mtd><mml:mn>3.343</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1.000</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>1.000</mml:mn></mml:mtd><mml:mtd><mml:mn>1.000</mml:mn></mml:mtd><mml:mtd><mml:mn>0.000</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0.000</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>25.000</mml:mn></mml:mtd><mml:mtd><mml:mn>0.000</mml:mn></mml:mtd><mml:mtd><mml:mn>0.000</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:mrow><mml:mtext>NE</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mn>0.0001.</mml:mn></mml:math></disp-formula>
<disp-formula id="eqn-35"><label>(35)</label><mml:math id="mml-eqn-35" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>GD</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mtable columnalign="left left left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>4.437</mml:mn></mml:mtd><mml:mtd><mml:mn>15.580</mml:mn></mml:mtd><mml:mtd><mml:mn>0.001</mml:mn></mml:mtd><mml:mtd><mml:mn>3.327</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0.997</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>0.997</mml:mn></mml:mtd><mml:mtd><mml:mn>1.000</mml:mn></mml:mtd><mml:mtd><mml:mn>0.002</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0.0202</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>24.987</mml:mn></mml:mtd><mml:mtd><mml:mn>0.000</mml:mn></mml:mtd><mml:mtd><mml:mn>0.0151</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:mrow><mml:mtext>NE</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mn>0.0013.</mml:mn></mml:math></disp-formula>
<disp-formula id="eqn-36"><label>(36)</label><mml:math id="mml-eqn-36" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>GD-QR</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mtable columnalign="left left left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>3.3124</mml:mn></mml:mtd><mml:mtd><mml:mn>14.3232</mml:mn></mml:mtd><mml:mtd><mml:mn>0.0969</mml:mn></mml:mtd><mml:mtd><mml:mn>2.5913</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0.8175</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>0.8712</mml:mn></mml:mtd><mml:mtd><mml:mn>0.9908</mml:mn></mml:mtd><mml:mtd><mml:mn>0.1169</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>1.0477</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>23.4959</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>0.3202</mml:mn></mml:mtd><mml:mtd><mml:mn>0.5065</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:mrow><mml:mtext>NE</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mn>0.0419.</mml:mn></mml:math></disp-formula></p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Convergence of GD, GD-QR and GD-TDR parameters.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-7.tif"/>
</fig><fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Norm of the parameter-estimation error.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-8.tif"/>
</fig>
</sec>
<sec id="s6_3">
<label>6.3</label>
<title>Discussion</title>
<p>GD and GD-TDR algorithms converge to high-accuracy estimates; however, GD-TDR exhibits markedly faster convergence and substantially shorter stagnation transients. In the considered experiment, GD requires approximately <inline-formula id="ieqn-127"><mml:math id="mml-ieqn-127"><mml:mn>800</mml:mn><mml:mspace width="thinmathspace" /><mml:mtext>s</mml:mtext></mml:math></inline-formula> to reach steady convergence, whereas GD-TDR attains comparable accuracy in about <inline-formula id="ieqn-128"><mml:math id="mml-ieqn-128"><mml:mn>200</mml:mn><mml:mspace width="thinmathspace" /><mml:mtext>s</mml:mtext></mml:math></inline-formula>. The error norm (NE) trajectories further indicate that GD spends a significant portion of the runtime in plateau-like regions before converging, a behavior consistent with ill-conditioning and the presence of flat directions in the quadratic objective function.</p>
<p>GD-QR mitigates oscillations during the convergence process but yields the poorest parameter convergence (<inline-formula id="ieqn-129"><mml:math id="mml-ieqn-129"><mml:mtext>NE</mml:mtext><mml:mo>=</mml:mo><mml:mn>0.0419</mml:mn></mml:math></inline-formula>). This behavior is expected, since the QR-based transformation effectively shifts the original optimal point to an alternative one in order to increase the convexity of the cost functional, thereby smoothing the convergence dynamics at the expense of final parameter accuracy.</p>
<p>From an algorithmic viewpoint, these improvements can be interpreted through Theorem 1: the decaying regularization increases the curvature of <inline-formula id="ieqn-130"><mml:math id="mml-ieqn-130"><mml:msub><mml:mi>J</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula> in the early iterations, preventing the gradient from vanishing far from the minimizer and discouraging weight growth in nearly-flat directions. As <inline-formula id="ieqn-131"><mml:math id="mml-ieqn-131"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> decreases, the method smoothly transitions toward the unregularized least-squares objective, thereby avoiding the steady-state bias that would occur if a fixed <inline-formula id="ieqn-132"><mml:math id="mml-ieqn-132"><mml:mi>&#x03B3;</mml:mi></mml:math></inline-formula> were used (GD-QR).</p>
</sec>
</sec>
<sec id="s7">
<label>7</label>
<title>Embedded Implementation on a Microcontroller</title>
<p>To evaluate suitability for low-processing-capacity platforms, GD-TDR was implemented on an &#x00AE; STM32F4-Nucleo microcontroller. The implementation was developed in the &#x00AE; Matlab-Simulink environment and deployed using the &#x00AE; Waijung toolkit. <xref ref-type="fig" rid="fig-9">Figs. 9</xref> and <xref ref-type="fig" rid="fig-10">10</xref> show the board-level program configuration, which follows the same signal flow as the numerical setup and adds serial communication blocks for monitoring. In particular, in <xref ref-type="fig" rid="fig-9">Fig. 9</xref>, the content of the block called chua can be located in <xref ref-type="fig" rid="fig-12">Fig. A1</xref> as well as its programming, which are contained in the <xref ref-type="app" rid="app-1">Appendix A</xref>.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>Program configuration for the &#x00AE; STM32F4-Nucleo board.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-9.tif"/>
</fig><fig id="fig-10">
<label>Figure 10</label>
<caption>
<title>CPU monitoring program.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-10.tif"/>
</fig>
<p><italic><bold>Experimental Results</bold></italic></p>
<p>In this work, neuron-paralysis-like behavior is quantitatively assessed through a combination of indicators, including prolonged plateaus in the loss evolution, persistent attenuation of the effective gradient norm, and excessive transient growth of the parameter vector prior to convergence.</p>
<p><xref ref-type="fig" rid="fig-11">Fig. 11</xref> reports the convergence behavior observed on the microcontroller. The results are qualitatively consistent with the numerical simulations, indicating that the proposed method preserves its robustness against stagnation even under limited precision and memory.
<disp-formula id="eqn-37"><label>(37)</label><mml:math id="mml-eqn-37" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">W</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>STM32</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mtable columnalign="left left left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>4.439</mml:mn></mml:mtd><mml:mtd><mml:mn>15.585</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>0.001</mml:mn></mml:mtd><mml:mtd><mml:mn>3.333</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0.997</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>0.987</mml:mn></mml:mtd><mml:mtd><mml:mn>1.000</mml:mn></mml:mtd><mml:mtd><mml:mn>0.001</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0.014</mml:mn></mml:mtd><mml:mtd><mml:mo>&#x2212;</mml:mo><mml:mn>24.988</mml:mn></mml:mtd><mml:mtd><mml:mn>0.001</mml:mn></mml:mtd><mml:mtd><mml:mn>0.007</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="2em" /><mml:mrow><mml:mtext>NE</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mn>0.001.</mml:mn></mml:math></disp-formula></p>
<fig id="fig-11">
<label>Figure 11</label>
<caption>
<title>Dynamics of convergence of GD-TDR weights on the &#x00AE; STM32F4-Nucleo board.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-11.tif"/>
</fig>
</sec>
<sec id="s8">
<label>8</label>
<title>Conclusions and Future Work</title>
<p>This paper introduced GD-TDR, a time-decaying quadratically regularized gradient-descent algorithm for training single-layer linear ANNs. The method is motivated by online identification problems in which the regressor data can be ill-conditioned and standard gradient descent may suffer from long plateaus, large intermediate weights, and neural-paralysis-like stagnation on finite-precision hardware. GD-TDR addresses this issue by enforcing strong convexity early in training through a quadratic penalty and then removing the penalty asymptotically via an exponential decay schedule.</p>
<p>A convergence theorem was provided that formalizes the key mechanism: for every iteration index the regularized objective remains strongly convex, so flat directions are eliminated, and under standard step-size conditions the iterates converge to the minimizer of the original (unregularized) least-squares cost as the regularization vanishes. The numerical validation on online identification of a Chua-type chaotic oscillator and the implementation on an &#x00AE; STM32F4-Nucleo microcontroller confirm that the proposed scheme converges faster than conventional gradient descent and significantly reduces stagnation transients, while preserving high identification accuracy.</p>
<p>Future work will focus on three directions. First, the decay schedule <inline-formula id="ieqn-133"><mml:math id="mml-ieqn-133"><mml:mi>&#x03B3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> can be adapted online using measurable indicators of conditioning or excitation, rather than being fixed <italic>a priori</italic>. Second, systematic and low-cost design rules for selecting <inline-formula id="ieqn-134"><mml:math id="mml-ieqn-134"><mml:mrow><mml:mtext mathvariant="bold">P</mml:mtext></mml:mrow></mml:math></inline-formula> (e.g., diagonal or structured choices compatible with embedded computation) will be investigated, including robustness to noise and time-varying parameters. Third, extending the approach beyond linear activation to shallow nonlinear networks and to constrained identification problems is of interest, where time-decaying regularization may provide similar benefits without sacrificing asymptotic accuracy.</p>
</sec>
</body>
<back>
<ack>
<p>The authors would like to thank Professor Alexander Poznyak for his valuable review of the work, as well as Bruce Dickinson for his motivation throughout the development of this research. They also acknowledge and are grateful for the funding provided by the IPN-SIP (SIP 20250023, 20250424, 20251300, 20251721, and 20253411), SECIHTI (CF-2023-I-1635), and the Sistema Nacional de Investigadores e Investigadoras (SNII) of Mexico.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>Funding was provided by the IPN-SIP (SIP 20250023, 20250424, 20251300, 20251721, 20253411 and MULTI-2026-0035), SECIHTI (CF-2023-I-1635), and the Sistema Nacional de Investigadores e Investigadoras (SNII) of Mexico.</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>Conceptualization, Sergio Isai Palomino-Resendiz and C&#x00E9;sar Ulises Sol&#x00ED;s-Cervantes; methodology, Diego Alonso Flores-Hern&#x00E1;ndez and Sergio Isai Palomino-Resendiz; software, C&#x00E9;sar Ulises Sol&#x00ED;s-Cervantes, Sergio Isai Palomino-Resendiz and Luis Alberto Cantera-Cantera; validation, Luis Alberto Cantera-Cantera and Jorge de Jes&#x00FA;s Morales-Mercado; formal analysis, C&#x00E9;sar Ulises Sol&#x00ED;s-Cervantes, Sergio Isai Palomino-Resendiz and Diego Alonso Flores-Hern&#x00E1;ndez; investigation, Sergio Isai Palomino-Resendiz; resources, C&#x00E9;sar Ulises Sol&#x00ED;s-Cervantes, Sergio Isai Palomino-Resendiz and Diego Alonso Flores-Hern&#x00E1;ndez; data curation, Luis Alberto Cantera-Cantera and Jorge de Jes&#x00FA;s Morales-Mercado; writing&#x2014;original draft preparation, Sergio Isai Palomino-Resendiz; writing&#x2014;review and editing, Sergio Isai Palomino-Resendiz and C&#x00E9;sar Ulises Sol&#x00ED;s-Cervantes; visualization, Sergio Isai Palomino-Resendiz; supervision, Sergio Isai Palomino-Resendiz and C&#x00E9;sar Ulises Sol&#x00ED;s-Cervantes; project administration, Sergio Isai Palomino-Resendiz; funding acquisition, Diego Alonso Flores-Hern&#x00E1;ndez and Sergio Isai Palomino-Resendiz. All authors reviewed and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>The data that support the findings of this study are available from the Corresponding Author, C&#x00E9;sar Ulises Sol&#x00ED;s-Cervantes, upon reasonable request.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest.</p>
</sec>
<glossary content-type="abbreviations" id="glossary-1">
<title>Abbreviations</title>
<def-list>
<def-item>
<term>The following abbreviations are used in this manuscript:</term>
</def-item>
<def-item>
<term>ANN</term>
<def>
<p>Artificial Neural Network</p>
</def>
</def-item>
<def-item>
<term>GD</term>
<def>
<p>Gradient Descent</p>
</def>
</def-item>
<def-item>
<term>GD-QR</term>
<def>
<p>Gradient Descent with Quadratic Regularization</p>
</def>
</def-item>
<def-item>
<term>GD-TDR</term>
<def>
<p>Gradient Descent with Time-Decaying Regularization</p>
</def>
</def-item>
<def-item>
<term>NP</term>
<def>
<p>Neural Paralysis</p>
</def>
</def-item>
<def-item>
<term>SLM</term>
<def>
<p>Stagnation in Local Minima</p>
</def>
</def-item>
</def-list>
</glossary>
<app-group id="appg-1">
<app id="app-1">
<title>Appendix A Chua Model Block</title>
<p>The following <sc>matlab</sc> function block implements <xref ref-type="disp-formula" rid="eqn-31">(31)</xref>.</p>
<p><monospace>function [xp, yp, zp] &#x003D; fcn(x,y,z)</monospace></p>
<p><monospace>alpha &#x003D; 15.6; beta &#x003D; 25; m0 &#x003D; &#x2212;8/7; m1 &#x003D; &#x2212;5/7;</monospace></p>
<p><monospace>phi &#x003D; m1&#x002A;x &#x002B; 0.5&#x002A;(m0-m1)&#x002A;(abs(x &#x002B; 1)-abs(x &#x2212; 1));</monospace></p>
<p><monospace>d1 &#x003D; y &#x2212; x &#x2212; phi; d2 &#x003D; x &#x2212; y &#x002B; z; d3 &#x003D; &#x2212;y;</monospace></p>
<p><monospace>xp &#x003D; alpha&#x002A;d1; yp &#x003D; d2; zp &#x003D; beta&#x002A;d3;</monospace></p>
<p><monospace>end</monospace></p>
<fig id="fig-12">
<label>Figure A1</label>
<caption>
<title>Contents of the block called <monospace>Chua</monospace> of the main program.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_77726-fig-12.tif"/>
</fig>
</app>
</app-group>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Pillonetto</surname> <given-names>G</given-names></string-name>, <string-name><surname>Aravkin</surname> <given-names>A</given-names></string-name>, <string-name><surname>Gedon</surname> <given-names>D</given-names></string-name>, <string-name><surname>Ljung</surname> <given-names>L</given-names></string-name>, <string-name><surname>Ribeiro</surname> <given-names>AH</given-names></string-name>, <string-name><surname>Sch&#x00F6;n</surname> <given-names>TB</given-names></string-name></person-group>. <article-title>Deep networks for system identification: a survey</article-title>. <source>Automatica</source>. <year>2025</year>;<volume>171</volume>(<issue>7</issue>):<fpage>111907</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.automatica.2024.111907</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dong</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>P</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Neural network-based parametric system identification: a comprehensive review</article-title>. <source>Int J Syst Sci</source>. <year>2023</year>;<volume>54</volume>(<issue>13</issue>):<fpage>2676</fpage>&#x2013;<lpage>88</lpage>. doi:<pub-id pub-id-type="doi">10.1080/00207721.2023.2241957</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>B</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>C</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Review on system identification, control, and optimization based on artificial intelligence</article-title>. <source>Mathematics</source>. <year>2025</year>;<volume>13</volume>(<issue>6</issue>):<fpage>952</fpage>. doi:<pub-id pub-id-type="doi">10.3390/math13060952</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Prince</surname> <given-names>SJD</given-names></string-name></person-group>. <source>Understanding deep learning</source>. <publisher-loc>Cambridge, MA, USA</publisher-loc>: <publisher-name>MIT Press</publisher-name>; <year>2023</year>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Bishop</surname> <given-names>CM</given-names></string-name>, <string-name><surname>Bishop</surname> <given-names>H</given-names></string-name></person-group>. <source>Deep learning: foundations and concepts</source>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name>; <year>2024</year>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Murphy</surname> <given-names>KP</given-names></string-name></person-group>. <source>Probabilistic machine learning: an introduction</source>. <publisher-loc>Cambridge, MA, USA</publisher-loc>: <publisher-name>MIT Press</publisher-name>; <year>2022</year>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Braun</surname> <given-names>L</given-names></string-name>, <string-name><surname>Domin&#x00E9;</surname> <given-names>CCJ</given-names></string-name>, <string-name><surname>Fitzgerald</surname> <given-names>JE</given-names></string-name>, <string-name><surname>Saxe</surname> <given-names>AM</given-names></string-name></person-group>. <article-title>Exact learning dynamics of deep linear networks with prior knowledge</article-title>. In: <conf-name>Proceedings of the 36th Conference on Neural Information Processing Systems (NeurIPS 2022)</conf-name>. <publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>; <year>2022</year>. p. <fpage>6615</fpage>&#x2013;<lpage>29</lpage>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ziyin</surname> <given-names>L</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Meng</surname> <given-names>X</given-names></string-name></person-group>. <article-title>Exact solutions of a deep linear network</article-title>. In: <conf-name>Proceedings of the 36th Conference on Neural Information Processing Systems (NeurIPS 2022)</conf-name>. <publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>; <year>2022</year>.p. <fpage>24446</fpage>&#x2013;<lpage>58</lpage>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gambella</surname> <given-names>C</given-names></string-name>, <string-name><surname>Ghaddar</surname> <given-names>B</given-names></string-name>, <string-name><surname>Naoum-Sawaya</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Optimization problems for machine learning: a survey</article-title>. <source>Eur J Oper Res</source>. <year>2021</year>;<volume>290</volume>(<issue>3</issue>):<fpage>807</fpage>&#x2013;<lpage>28</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.ejor.2020.08.045</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ahn</surname> <given-names>K</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Sra</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Understanding the unstable convergence of gradient descent</article-title>. In: <conf-name>Proceedings of the 39th International Conference on Machine Learning (ICML)</conf-name>. <publisher-loc>London, UK</publisher-loc>: <publisher-name>PMLR</publisher-name>; <year>2022</year>. p. <fpage>247</fpage>&#x2013;<lpage>57</lpage>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Xie</surname> <given-names>J</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>R</given-names></string-name>, <string-name><surname>Nijkamp</surname> <given-names>E</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>S-C</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>YN</given-names></string-name></person-group>. <article-title>Cooperative training of fast thinking initializer and slow thinking solver for conditional learning</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. <year>2022</year>;<volume>44</volume>(<issue>8</issue>):<fpage>3957</fpage>&#x2013;<lpage>73</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TPAMI.2021.3069023</pub-id>; <pub-id pub-id-type="pmid">33769930</pub-id></mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wurzberger</surname> <given-names>J</given-names></string-name>, <string-name><surname>Schwenker</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Learning in deep radial basis function networks</article-title>. <source>Entropy</source>. <year>2024</year>;<volume>26</volume>(<issue>5</issue>):<fpage>368</fpage>. doi:<pub-id pub-id-type="doi">10.3390/e26050368</pub-id>; <pub-id pub-id-type="pmid">38785617</pub-id></mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kalina</surname> <given-names>J</given-names></string-name>, <string-name><surname>Vidnerov&#x00E1;</surname> <given-names>P</given-names></string-name>, <string-name><surname>Jan&#x00E1;&#x010D;ek</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Highly robust training of regularized radial basis function networks</article-title>. <source>Kybernetika</source>. <year>2024</year>;<volume>60</volume>(<issue>1</issue>):<fpage>38</fpage>&#x2013;<lpage>59</lpage>. doi:<pub-id pub-id-type="doi">10.14736/kyb-2024-1-0038</pub-id>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>N</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>D</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lewis</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Recent progress in reinforcement learning and adaptive dynamic programming for advanced control applications</article-title>. <source>IEEE/CAA J Autom Sin</source>. <year>2024</year>;<volume>11</volume>(<issue>1</issue>):<fpage>18</fpage>&#x2013;<lpage>36</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JAS.2023.123843</pub-id>; <pub-id pub-id-type="pmid">25079929</pub-id></mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Le</surname> <given-names>T-L</given-names></string-name>, <string-name><surname>Huynh</surname> <given-names>T-T</given-names></string-name>, <string-name><surname>Hong</surname> <given-names>S-K</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>C-M</given-names></string-name></person-group>. <article-title>Hybrid neural network cerebellar model articulation controller design for non-linear dynamic time-varying plants</article-title>. <source>Front Neurosci</source>. <year>2020</year>;<volume>14</volume>:<fpage>695</fpage>. doi:<pub-id pub-id-type="doi">10.3389/fnins.2020.00695</pub-id>; <pub-id pub-id-type="pmid">32848536</pub-id></mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Razmi</surname> <given-names>M</given-names></string-name>, <string-name><surname>Macnab</surname> <given-names>CJB</given-names></string-name></person-group>. <article-title>Near-optimal neural-network robot control with adaptive gravity compensation</article-title>. <source>Neurocomputing</source>. <year>2020</year>;<volume>389</volume>(<issue>6</issue>):<fpage>83</fpage>&#x2013;<lpage>92</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.neucom.2020.01.026</pub-id>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tian</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Recent advances in stochastic gradient descent in deep learning</article-title>. <source>Mathematics</source>. <year>2023</year>;<volume>11</volume>(<issue>3</issue>):<fpage>682</fpage>. doi:<pub-id pub-id-type="doi">10.3390/math11030682</pub-id>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>W</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>C</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Harnessing machine learning for identifying parameters in fractional chaotic systems</article-title>. <source>Appl Math Comput</source>. <year>2025</year>;<volume>500</volume>(<issue>2</issue>):<fpage>129454</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.amc.2025.129454</pub-id>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chakraborty</surname> <given-names>S</given-names></string-name>, <string-name><surname>Gr&#x00F6;&#x00DF;mann</surname> <given-names>AH</given-names></string-name>, <string-name><surname>Benner</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Divide and conquer: learning chaotic dynamical systems using deep neural networks</article-title>. <source>Comput Methods Appl Mech Eng</source>. <year>2024</year>;<volume>430</volume>(<issue>8</issue>):<fpage>117442</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.cma.2024.117442</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Mikhaeil</surname> <given-names>JM</given-names></string-name>, <string-name><surname>Monfared</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Durstewitz</surname> <given-names>D</given-names></string-name></person-group>. <article-title>On the difficulty of learning chaotic dynamics with RNNs</article-title>. In: <conf-name>NIPS&#x2019;22: Proceedings of the 36th International Conference on Neural Information Processing Systems</conf-name>. <publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>; <year>2022</year>. p. <fpage>11297</fpage>&#x2013;<lpage>312</lpage>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mohan</surname> <given-names>N</given-names></string-name>, <string-name><surname>Hosni</surname> <given-names>A</given-names></string-name>, <string-name><surname>Atef</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Neural networks implementations on fpga for biomedical applications: a review</article-title>. <source>SN Computer Sci</source>. <year>2024</year>;<volume>5</volume>(<issue>8</issue>):<fpage>1004</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s42979-024-03381-4</pub-id>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Majumdar</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Spiking neural networks: a comprehensive review of diverse applications, research progress, challenges and future research directions</article-title>. <source>Evol Syst</source>. <year>2025</year>;<volume>16</volume>(<issue>4</issue>):<fpage>125</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s12530-025-09755-0</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Achour</surname> <given-names>EM</given-names></string-name>, <string-name><surname>Malgouyres</surname> <given-names>F</given-names></string-name>, <string-name><surname>Gerchinovitz</surname> <given-names>S</given-names></string-name></person-group>. <article-title>The loss landscape of deep linear neural networks: a second-order analysis</article-title>. <source>J Mach Learn Res</source>. <year>2024</year>;<volume>25</volume>:<fpage>242</fpage>.</mixed-citation></ref>
</ref-list>
</back></article>


