<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="review-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMES</journal-id>
<journal-id journal-id-type="nlm-ta">CMES</journal-id>
<journal-id journal-id-type="publisher-id">CMES</journal-id>
<journal-title-group>
<journal-title>Computer Modeling in Engineering &#x0026; Sciences</journal-title>
</journal-title-group>
<issn pub-type="epub">1526-1506</issn>
<issn pub-type="ppub">1526-1492</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">80601</article-id>
<article-id pub-id-type="doi">10.32604/cmes.2026.080601</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Review</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>From Lexicons to Large Language Models: A Comprehensive Survey of Sentiment Analysis Methods, Benchmarks, and Emerging Frontiers</article-title>
<alt-title alt-title-type="left-running-head">From Lexicons to Large Language Models: A Comprehensive Survey of Sentiment Analysis Methods, Benchmarks, and Emerging Frontiers</alt-title>
<alt-title alt-title-type="right-running-head">From Lexicons to Large Language Models: A Comprehensive Survey of Sentiment Analysis Methods, Benchmarks, and Emerging Frontiers</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author" corresp="yes">
<name name-style="western"><surname>De</surname><given-names>Shuvodeep</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><email>vvg26@txstate.edu</email></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Gosai</surname><given-names>Agnivo</given-names></name><xref ref-type="aff" rid="aff-2">2</xref><xref ref-type="author-notes" rid="afn1">#</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Thankachan</surname><given-names>Karun</given-names></name><xref ref-type="aff" rid="aff-3">3</xref><xref ref-type="author-notes" rid="afn1">#</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>ZeinEldin</surname><given-names>Ramadan A.</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Almaktoom</surname><given-names>Abdulaziz T.</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Bayram</surname><given-names>Mustafa</given-names></name><xref ref-type="aff" rid="aff-6">6</xref></contrib>
<contrib id="author-7" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Mohamed</surname><given-names>Ali Wagdy</given-names></name><xref ref-type="aff" rid="aff-7">7</xref><xref ref-type="aff" rid="aff-8">8</xref><email>aliwagdy@gmail.com</email></contrib>
<aff id="aff-1"><label>1</label><institution>Ingram School of Engineering, Texas State University</institution>, <addr-line>San Marcos, TX</addr-line>, <country>USA</country></aff>
<aff id="aff-2"><label>2</label><institution>Corning Incorporated</institution>, <addr-line>Painted Post, NY</addr-line>, <country>USA</country></aff>
<aff id="aff-3"><label>3</label><institution>Language Technologies Institute, School of Computer Science (SCS), Carnegie Mellon University</institution>, <addr-line>Pittsburgh, PA</addr-line>, <country>USA</country></aff>
<aff id="aff-4"><label>4</label><institution>Deanship of Scientific Research, King Abdulaziz University</institution>, <addr-line>Jeddah</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-5"><label>5</label><institution>Department of Operations and Supply Chain Management, Effat University</institution>, <addr-line>Jeddah</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-6"><label>6</label><institution>Department of Computer Engineering, Biruni University</institution>, <addr-line>Istanbul</addr-line>, <country>Turkey</country></aff>
<aff id="aff-7"><label>7</label><institution>Operations Research Department, Faculty of Graduate Studies for Statistical Research, Cairo University</institution>, <addr-line>Giza</addr-line>, <country>Egypt</country></aff>
<aff id="aff-8"><label>8</label><institution>School of Business, University of Science and Technology, Zewail City of Science and Technology</institution>, <addr-line>6th of October City, Giza</addr-line>, <country>Egypt</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Authors: Shuvodeep De. Email: <email>vvg26@txstate.edu</email>; Ali Wagdy Mohamed. Email: <email>aliwagdy@gmail.com</email></corresp>
<fn id="afn1">
<p><sup>#</sup>These authors contributed equally to this work</p>
</fn>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2026</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>27</day><month>5</month><year>2026</year>
</pub-date>
<volume>147</volume>
<issue>2</issue>
<elocation-id>5</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>02</month>
<year>2026</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>04</month>
<year>2026</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2026 The Authors. Published by Tech Science Press.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>The Authors</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMES_80601.pdf"></self-uri>
<abstract>
<p>Sentiment analysis (SA) has evolved from a niche text-classification task into a central problem in natural language processing, spanning multiple domains, modalities, and languages. This survey provides a comprehensive review of sentiment analysis methods from their origins in lexicon-based approaches through classical machine learning, deep learning architectures, pre-trained transformers, and the current era of large language models (LLMs). We formalize the SA problem across multiple granularity levels (document, sentence, and aspect) and present a taxonomy that encompasses classification, regression, aspect-based sentiment analysis (ABSA), emotion detection, and stance detection tasks across diverse domains including movie reviews, product reviews, healthcare, finance, and social media. We review benchmark datasets spanning text-only corpora (IMDb, SST, SemEval series), multimodal benchmarks (CMU-MOSI, CMU-MOSEI, MELD), and domain-specific evaluation suites such as SentiEval. The methodological evolution is traced from VADER and SentiWordNet, through SVM and Na&#x00EF;ve Bayes classifiers, CNN and LSTM architectures, BERT and its variants, to modern LLMs including GPT-4, Llama 3, and ModernBERT, with technical details of key architectures and their mathematical formulations. We provide dedicated analyses of chain-of-thought reasoning for implicit sentiment, multimodal fusion strategies, cross-lingual transfer methods, sarcasm and irony detection, explainability through SHAP and LIME, and the emerging challenge of AI-generated fake reviews. A comparative analysis across paradigms reveals that while LLMs achieve strong zero-shot performance, fine-tuned smaller models remain competitive on standard benchmarks, a finding with significant implications for deployment efficiency. We identify persistent open challenges including domain drift, cultural bias, and the model variability problem, and outline future research directions encompassing reasoning-augmented SA, agentic workflows, federated learning, and real-time edge deployment. With coverage of over 130 references spanning two decades of research and 29 new references from 2024 and 2025, this survey provides a unified roadmap for both newcomers and researchers at the frontier of sentiment analysis.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Sentiment analysis</kwd>
<kwd>opinion mining</kwd>
<kwd>large language models</kwd>
<kwd>transformers</kwd>
<kwd>BERT</kwd>
<kwd>aspect-based sentiment analysis</kwd>
<kwd>multimodal sentiment analysis</kwd>
<kwd>cross-lingual NLP</kwd>
<kwd>explainable AI</kwd>
<kwd>chain-of-thought reasoning</kwd>
<kwd>sarcasm detection</kwd>
<kwd>benchmark datasets</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>Deanship of Scientific Research (DSR) at King Abdulaziz University</funding-source>
<award-id>IPP: 543-305-2025</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Sentiment analysis, the computational task of identifying and extracting subjective information from text, has been a foundational problem in natural language processing (NLP) for over two decades [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-2">2</xref>]. What began with the pioneering 2002 work of Pang, Lee, and Vaithyanathan, who applied machine learning to movie review classification [<xref ref-type="bibr" rid="ref-1">1</xref>], has since grown into a large and widely used research and application area that touches nearly every domain where human opinion matters, including product reviews, healthcare monitoring, financial forecasting, political discourse analysis, and social media monitoring [<xref ref-type="bibr" rid="ref-3">3</xref>&#x2013;<xref ref-type="bibr" rid="ref-5">5</xref>]. Over the course of its development, the field has undergone several distinct paradigm shifts, each building upon and eventually superseding the limitations of its predecessors.</p>
<p>The earliest systematic approaches relied on sentiment lexicons and rule-based resources such as SentiWordNet [<xref ref-type="bibr" rid="ref-6">6</xref>,<xref ref-type="bibr" rid="ref-7">7</xref>] and VADER [<xref ref-type="bibr" rid="ref-8">8</xref>], which assigned polarity scores to individual words or phrases. While interpretable and resource-efficient, these methods could not account for context-dependent sentiment, compositional semantics, or figurative language. Classical machine learning methods, including support vector machines (SVMs) [<xref ref-type="bibr" rid="ref-9">9</xref>], Na&#x00EF;ve Bayes classifiers [<xref ref-type="bibr" rid="ref-10">10</xref>], and logistic regression, marked the first major advance by learning discriminative features directly from labeled corpora [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-11">11</xref>], achieving substantially higher accuracy on benchmark tasks. The subsequent deep learning revolution introduced convolutional neural networks (CNNs) [<xref ref-type="bibr" rid="ref-12">12</xref>], recurrent architectures including long short-term memory (LSTM) networks [<xref ref-type="bibr" rid="ref-13">13</xref>], and attention mechanisms [<xref ref-type="bibr" rid="ref-14">14</xref>,<xref ref-type="bibr" rid="ref-15">15</xref>], all of which could capture sequential dependencies and long-range context without manual feature engineering. A more fundamental shift arrived with transfer learning through pre-trained transformers such as BERT [<xref ref-type="bibr" rid="ref-16">16</xref>], RoBERTa [<xref ref-type="bibr" rid="ref-17">17</xref>], and XLNet [<xref ref-type="bibr" rid="ref-18">18</xref>], which changed the landscape by enabling fine-tuning on downstream sentiment tasks with minimal labeled data [<xref ref-type="bibr" rid="ref-19">19</xref>]. Most recently, large language models (LLMs) such as GPT-3 [<xref ref-type="bibr" rid="ref-20">20</xref>], GPT-4-class models, Llama 3 [<xref ref-type="bibr" rid="ref-21">21</xref>], and their instruction-tuned variants have shown strong zero-shot and few-shot sentiment analysis performance on a range of benchmarks [<xref ref-type="bibr" rid="ref-22">22</xref>&#x2013;<xref ref-type="bibr" rid="ref-24">24</xref>], while simultaneously raising new questions about reliability, cost, and interpretability [<xref ref-type="bibr" rid="ref-25">25</xref>]. <xref ref-type="fig" rid="fig-1">Fig. 1</xref> provides a high-level overview of this methodological evolution, situating the major sentiment analysis paradigms that frame the structure of the remainder of this survey.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Evolution of sentiment analysis methodologies from early lexicon-based approaches to modern large language models. The timeline highlights major paradigm shifts, representative model families, and their characteristic strengths and limitations (This figure was generated with the assistance of OpenAI GPT 5.2).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-1.tif"/>
</fig>
<p>Despite two decades of sustained progress, sentiment analysis remains far from solved. Sarcasm and irony continue to confound even state-of-the-art systems [<xref ref-type="bibr" rid="ref-26">26</xref>&#x2013;<xref ref-type="bibr" rid="ref-28">28</xref>], because the intended meaning is often the polar opposite of what is literally expressed. Domain and temporal drift remain persistent sources of performance degradation, as models trained in one context, whether a product category or a time period, frequently fail to generalize to others [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-30">30</xref>]. The challenge of cross-lingual transfer to low-resource languages, where labeled training data is scarce or nonexistent, has attracted growing attention but remains largely unsolved [<xref ref-type="bibr" rid="ref-31">31</xref>&#x2013;<xref ref-type="bibr" rid="ref-33">33</xref>]. Multimodal fusion across text, audio, and video introduces additional complexity, requiring models to integrate fundamentally different signal types into a coherent sentiment judgment [<xref ref-type="bibr" rid="ref-34">34</xref>&#x2013;<xref ref-type="bibr" rid="ref-36">36</xref>]. Furthermore, the proliferation of LLMs has given rise to an entirely new threat: the generation of sophisticated AI-generated fake reviews that can evade traditional detection systems [<xref ref-type="bibr" rid="ref-37">37</xref>,<xref ref-type="bibr" rid="ref-38">38</xref>]. Adding to these concerns, recent research has identified a <italic>model variability problem</italic> (MVP) in LLM-based sentiment analysis, whereby identical inputs can produce inconsistent classifications across runs due to stochastic inference and prompt sensitivity [<xref ref-type="bibr" rid="ref-25">25</xref>], a finding with troubling implications for deployment in high-stakes applications.</p>
<p>This survey aims to provide a comprehensive and technically grounded review of the sentiment analysis landscape, from its origins to its current frontiers. Our approach is distinguished by five core contributions that collectively provide a unifying framework absent from prior surveys: (1) a formal taxonomy (<xref ref-type="sec" rid="s2">Section 2</xref>) that organizes sentiment analysis along orthogonal dimensions of granularity, task formulation, domain, and output space, serving as a scaffold that connects all subsequent methodological discussion; (2) mathematical formulations for key architectures across all five paradigms, enabling readers to understand not just <italic>what</italic> works but <italic>why</italic>; (3) coverage of 29 new references from 2024 and 2025; (4) substantive treatment of topics largely absent from existing surveys, including the model variability problem, agentic SA workflows, and AI-generated fake review detection; and (5) a structured practitioner decision framework for choosing among paradigms based on specific engineering constraints.</p>
<p>Several prior surveys have made valuable contributions to the systematic understanding of sentiment analysis. Wankhade et al. [<xref ref-type="bibr" rid="ref-4">4</xref>] surveyed core methods, applications, and challenges, while Jain et al. [<xref ref-type="bibr" rid="ref-39">39</xref>] conducted a systematic review of ML methods for consumer sentiment analysis in hospitality and tourism reviews. Raghunathan and Saravanakumar [<xref ref-type="bibr" rid="ref-40">40</xref>] cataloged persistent methodological challenges, and Tetteh and Thushara [<xref ref-type="bibr" rid="ref-41">41</xref>] focused specifically on evaluation tools for movie review sentiment analysis. Subsequent surveys expanded the scope to include deep learning and hybrid approaches, notably those by Birjali et al. [<xref ref-type="bibr" rid="ref-5">5</xref>], Islam et al. [<xref ref-type="bibr" rid="ref-42">42</xref>], Bordoloi and Biswas [<xref ref-type="bibr" rid="ref-43">43</xref>], and Mao et al. [<xref ref-type="bibr" rid="ref-44">44</xref>].</p>
<p>More recently, a new generation of surveys has emerged reflecting the rapid evolution of sentiment analysis in the post-transformer and generative AI era. Kumar et al. [<xref ref-type="bibr" rid="ref-45">45</xref>] provide a comprehensive review tracing the progression from classical machine learning techniques to transformer-based architectures, while Alahmadi et al. [<xref ref-type="bibr" rid="ref-46">46</xref>] examines generalization challenges and emerging trends across application domains. Suryawanshi [<xref ref-type="bibr" rid="ref-47">47</xref>] surveys machine learning and deep learning techniques with an emphasis on practical applications, and Bachate and Suchitra [<xref ref-type="bibr" rid="ref-48">48</xref>] focuses on sentiment and emotion recognition in social media contexts. In parallel, Krugmann and Hartmann [<xref ref-type="bibr" rid="ref-23">23</xref>] discusses the impact of generative AI on sentiment analysis workflows, and recent technical reports [<xref ref-type="bibr" rid="ref-49">49</xref>] systematically catalog datasets, tools, and evaluation challenges. Ahmad Alomari [<xref ref-type="bibr" rid="ref-50">50</xref>] presented a comprehensive PRISMA-guided survey evaluating ChatGPT across multiple NLP tasks, highlighting its adaptability, performance trends, and key limitations in real-world applications. Despite these contributions, existing surveys typically address large language models only in isolation or as an extension of prior paradigms, motivating the need for a unified survey spanning the full methodological evolution from lexicon-based methods to modern LLM-centric sentiment analysis.</p>
<p>However, the present survey distinguishes itself from these prior works in several critical respects. First, unlike surveys and modeling works limited to movie reviews [<xref ref-type="bibr" rid="ref-41">41</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>] or single methodological families, we span the complete pipeline from lexicons to reasoning-augmented LLMs across multiple domains, including movies, products, healthcare, finance, and social media. Second, we provide mathematical formulations for key architectures, from the SVM dual formulation and LSTM gating equations to the transformer self-attention mechanism and LoRA adaptation, enabling readers to understand not just <italic>what</italic> works but <italic>why</italic>. Third, we have studied and discussed several new references from 2024 and 2025, covering LLM benchmarking [<xref ref-type="bibr" rid="ref-22">22</xref>,<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-52">52</xref>,<xref ref-type="bibr" rid="ref-53">53</xref>], chain-of-thought reasoning for sentiment [<xref ref-type="bibr" rid="ref-54">54</xref>&#x2013;<xref ref-type="bibr" rid="ref-57">57</xref>], ModernBERT [<xref ref-type="bibr" rid="ref-58">58</xref>], multimodal LLMs [<xref ref-type="bibr" rid="ref-36">36</xref>,<xref ref-type="bibr" rid="ref-59">59</xref>,<xref ref-type="bibr" rid="ref-60">60</xref>], cross-lingual methods [<xref ref-type="bibr" rid="ref-33">33</xref>,<xref ref-type="bibr" rid="ref-61">61</xref>&#x2013;<xref ref-type="bibr" rid="ref-65">65</xref>], sarcasm detection with LLMs [<xref ref-type="bibr" rid="ref-28">28</xref>,<xref ref-type="bibr" rid="ref-66">66</xref>,<xref ref-type="bibr" rid="ref-67">67</xref>], explainable SA [<xref ref-type="bibr" rid="ref-68">68</xref>&#x2013;<xref ref-type="bibr" rid="ref-70">70</xref>], AI-generated fake reviews [<xref ref-type="bibr" rid="ref-37">37</xref>,<xref ref-type="bibr" rid="ref-38">38</xref>], and ensemble strategies [<xref ref-type="bibr" rid="ref-71">71</xref>]. Fourth, we dedicate substantive treatment to topics that are largely absent from existing surveys, including AI-generated fake review detection, the model variability problem, agentic SA workflows, and explainability.</p>
<p>The remainder of this paper is organized as follows. <xref ref-type="sec" rid="s1_1">Section 1.1</xref> describes our survey methodology. <xref ref-type="sec" rid="s2">Section 2</xref> formalizes the sentiment analysis problem and presents a taxonomy of tasks, granularity levels, and domains. <xref ref-type="sec" rid="s3">Section 3</xref> reviews benchmark datasets and evaluation metrics. <xref ref-type="sec" rid="s4">Section 4</xref> traces the methodological evolution from lexicons through LLMs, providing technical details at each stage. <xref ref-type="sec" rid="s5">Section 5</xref> examines domain-specific challenges including sarcasm detection, domain drift, and AI-generated fake reviews. <xref ref-type="sec" rid="s6">Section 6</xref> surveys emerging frontiers such as reasoning-augmented SA, agentic workflows, federated learning, and explainability. <xref ref-type="sec" rid="s7">Section 7</xref> identifies research gaps and future directions. Finally, <xref ref-type="sec" rid="s8">Section 8</xref> concludes the survey.</p>
<sec id="s1_1">
<label>1.1</label>
<title>Survey Methodology</title>
<p>To ensure comprehensive and reproducible coverage, this survey follows a structured narrative synthesis methodology. We searched five major academic databases: Scopus, Web of Science, IEEE Xplore, ACL Anthology, and Google Scholar. Primary search queries included combinations of the terms &#x201C;sentiment analysis,&#x201D; &#x201C;opinion mining,&#x201D; &#x201C;aspect-based sentiment analysis,&#x201D; &#x201C;large language models AND sentiment,&#x201D; &#x201C;multimodal sentiment,&#x201D; &#x201C;cross-lingual sentiment,&#x201D; &#x201C;sarcasm detection,&#x201D; and &#x201C;explainable sentiment analysis.&#x201D; Searches were conducted iteratively with special focus in the time-span: October 2024 and February 2026.</p>
<p>Inclusion criteria required that papers be: (1) published in peer-reviewed venues or established preprint servers (arXiv) with demonstrable community impact; (2) directly relevant to sentiment analysis methodology, evaluation, or application; and (3) available in English. We excluded: purely application-specific case studies without methodological contribution, duplicate publications, and works superseded by later versions from the same authors. For rapidly evolving topics (LLM evaluation, multimodal SA, cross-lingual transfer), we prioritized publications from 2024&#x2013;2025 to ensure currency.</p>
<p>Starting from an initial pool of over 300 candidate papers identified through database searches, we applied snowball sampling by tracing citation networks to identify seminal works and recent extensions. After applying inclusion/exclusion criteria and removing duplicates, the final survey covers 140 references covering the most recent developments. While this survey follows a narrative synthesis approach rather than a strict PRISMA systematic review protocol which is better suited to meta-analyses of empirical studies the structured search and selection process ensures breadth and reproducibility.</p>
</sec>
</sec>
<sec id="s2">
<label>2</label>
<title>Problem Formulation and Taxonomy</title>
<p>To ground this historical and methodological overview in a precise analytical framework, it is necessary to formalize what constitutes a sentiment analysis task and how different problem formulations relate to one another. While the term &#x201C;sentiment analysis&#x201D; is often used broadly, it encompasses a diverse family of tasks that vary in granularity, output space, and application context. The taxonomy presented in <xref ref-type="fig" rid="fig-2">Fig. 2</xref> serves as the unifying framework for this survey: each methodological paradigm discussed in <xref ref-type="sec" rid="s4">Section 4</xref> can be understood as a different parameterization of the formal problem defined here, and each domain-specific challenge in <xref ref-type="sec" rid="s5">Section 5</xref> corresponds to a specific failure mode within one or more dimensions of this taxonomy. The next section therefore introduces a formal problem definition and a unified taxonomy that will serve as a reference point for the methodological and empirical discussions that follow.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Taxonomy of sentiment analysis tasks. The diagram organizes sentiment analysis along orthogonal dimensions including granularity level, task formulation, extended task variants, application domains, and output spaces (This figure was generated with the assistance of OpenAI GPT 5.2).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-2.tif"/>
</fig>
<sec id="s2_1">
<label>2.1</label>
<title>Formal Problem Definition</title>
<p>Sentiment analysis most commonly classifies text into positive, negative, or neutral categories, although many benchmarks use fine-grained polarity scales such as five-class sentiment (very negative, negative, neutral, positive, very positive). Thus, we can formally define sentiment analysis can be formally defined as a mapping function <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mi>f</mml:mi><mml:mo>:</mml:mo><mml:mrow><mml:mi>&#x1D4B3;</mml:mi></mml:mrow><mml:mo stretchy="false">&#x2192;</mml:mo><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow></mml:math></inline-formula>, where <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mrow><mml:mi>&#x1D4B3;</mml:mi></mml:mrow></mml:math></inline-formula> denotes the input space (text, audio, video, or their combination) and <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow></mml:math></inline-formula> denotes the output space of sentiment labels or scores. The nature of <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow></mml:math></inline-formula> varies with the task formulation: in binary classification, <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mtext>positive</mml:mtext><mml:mo>,</mml:mo><mml:mtext>negative</mml:mtext><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula>; in ternary classification, a neutral class is added so that <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mtext>positive</mml:mtext><mml:mo>,</mml:mo><mml:mtext>neutral</mml:mtext><mml:mo>,</mml:mo><mml:mtext>negative</mml:mtext><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula> or a multi-class polarity scale; fine-grained settings expand this further to ordinal scales such as <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>4</mml:mn><mml:mo>,</mml:mo><mml:mn>5</mml:mn><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula>; regression formulations define continuous output spaces such as <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> or <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula>; and aspect-based formulations condition the mapping on a specific aspect <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mi>a</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi>&#x1D49C;</mml:mi></mml:mrow></mml:math></inline-formula>, yielding <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:mi>f</mml:mi><mml:mo>:</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x1D4B3;</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">&#x2192;</mml:mo><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow></mml:math></inline-formula>.</p>
<p>For a document <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:mi>d</mml:mi></mml:math></inline-formula> consisting of a sequence of tokens <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, the goal is to learn parameters <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi></mml:math></inline-formula> that maximize the conditional likelihood:<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msup><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:mi>arg</mml:mi><mml:mo>&#x2061;</mml:mo><mml:munder><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi></mml:mrow></mml:munder><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>;</mml:mo><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> denotes the input instance (e.g., a sequence of tokens representing a document or sentence), <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> is the corresponding sentiment label or score, and <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mi>&#x03B8;</mml:mi></mml:math></inline-formula> represents the model parameters and <italic>N</italic> is the number of training examples. The summation defines the log-likelihood objective, and <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup></mml:math></inline-formula> denotes the optimal model parameters that maximize this objective. The conditional probability <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>;</mml:mo><mml:mi>&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is parameterized differently depending on the model family, ranging from linear classifiers to deep neural networks. This objective corresponds to maximizing the likelihood of the observed training data under the model, i.e., encouraging the model to assign high probability to correct sentiment labels. Taking the logarithm converts the product of probabilities into a sum, improving numerical stability and simplifying optimization. This formulation also highlights a key tension that recurs throughout the survey: supervised methods optimize <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref> directly but require labeled data, while LLM-based approaches approximate the same mapping through in-context learning without explicit parameter optimization on task-specific data, trading annotation cost for computational cost and introducing the model variability problem discussed in <xref ref-type="sec" rid="s4_5_4">Section 4.5.4</xref>.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Granularity Levels</title>
<p>Sentiment analysis is typically studied across multiple granularity levels: document-level (overall polarity), sentence-level, target-level (sentiment toward specific entities), and aspect-level (polarity toward specific entity attributes), each progressively narrowing the analytical focus [<xref ref-type="bibr" rid="ref-3">3</xref>,<xref ref-type="bibr" rid="ref-4">4</xref>,<xref ref-type="bibr" rid="ref-72">72</xref>]. At the <bold>document level</bold>, a single sentiment label is assigned to an entire document such as a movie review [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-73">73</xref>]. This formulation assumes a single dominant opinion, an assumption that breaks down for documents discussing multiple entities or aspects. <bold>Sentence or Phrase-level</bold> SA addresses this limitation by classifying individual sentences, recognizing that a document may contain mixed sentiments [<xref ref-type="bibr" rid="ref-74">74</xref>,<xref ref-type="bibr" rid="ref-75">75</xref>]. Wilson et al. [<xref ref-type="bibr" rid="ref-74">74</xref>] demonstrated that contextual polarity at the phrase level often diverges from prior word-level polarity, underscoring the importance of finer-grained analysis.</p>
<p>The most detailed formulation is <bold>aspect-level</bold> SA, also known as aspect-based sentiment analysis (ABSA), identifies sentiment toward specific attributes or aspects of an entity mentioned in the text (e.g., food quality, service, price) [<xref ref-type="bibr" rid="ref-76">76</xref>&#x2013;<xref ref-type="bibr" rid="ref-80">80</xref>]. For example, in the sentence &#x201C;The cinematography was stunning but the plot was predictable,&#x201D; the aspect <italic>cinematography</italic> carries positive sentiment while <italic>plot</italic> carries negative sentiment. Formally, given input <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow></mml:math></inline-formula> and aspect term <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:mi>a</mml:mi></mml:math></inline-formula>, ABSA solves:<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>a</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>arg</mml:mi><mml:mo>&#x2061;</mml:mo><mml:munder><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow></mml:mrow></mml:munder><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo>;</mml:mo><mml:msup><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>here, <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow></mml:math></inline-formula> denotes the set of possible sentiment labels (e.g., positive, negative, neutral). At inference time, the trained parameters <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:msup><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo>&#x2217;</mml:mo></mml:msup></mml:math></inline-formula> are fixed and used to predict the most likely sentiment label.</p>
<p>Recent work has further extended this to sub-aspect analysis [<xref ref-type="bibr" rid="ref-79">79</xref>] and cross-lingual ABSA [<xref ref-type="bibr" rid="ref-33">33</xref>], where aspect-sentiment pairs must be identified across languages with limited target-language supervision. <xref ref-type="fig" rid="fig-3">Fig. 3</xref> illustrates how the same text can yield substantially different sentiment interpretations depending on the chosen granularity level.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Illustration of sentiment analysis granularity levels. The same input text yields different sentiment interpretations depending on whether sentiment is modeled at the document, sentence, or aspect level (This figure was generated with the assistance of OpenAI GPT 5.2).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-3.tif"/>
</fig>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Task Variants</title>
<p>Beyond polarity classification, the sentiment analysis ecosystem encompasses several related tasks [<xref ref-type="bibr" rid="ref-4">4</xref>,<xref ref-type="bibr" rid="ref-43">43</xref>,<xref ref-type="bibr" rid="ref-44">44</xref>]. <bold>Emotion detection</bold> extends binary polarity to multi-label emotion taxonomies such as Plutchik&#x2019;s wheel or the GoEmotions taxonomy with 27 emotion categories [<xref ref-type="bibr" rid="ref-81">81</xref>], where the output space becomes <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mn>2</mml:mn><mml:mrow><mml:mrow><mml:mi>&#x2130;</mml:mi></mml:mrow></mml:mrow></mml:msup></mml:math></inline-formula> with <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:mrow><mml:mi>&#x2130;</mml:mi></mml:mrow></mml:math></inline-formula> denoting the set of emotions. <bold>Stance detection</bold> determines the author&#x2019;s position (favor, against, neutral) toward a specific target, which may not be explicitly mentioned in the text [<xref ref-type="bibr" rid="ref-82">82</xref>]. <bold>Multimodal SA</bold> extends the input space to <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:mrow><mml:mi>&#x1D4B3;</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D4B3;</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D4B3;</mml:mi></mml:mrow><mml:mi>a</mml:mi></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D4B3;</mml:mi></mml:mrow><mml:mi>v</mml:mi></mml:msub></mml:math></inline-formula>, encompassing textual, acoustic, and visual modalities [<xref ref-type="bibr" rid="ref-34">34</xref>&#x2013;<xref ref-type="bibr" rid="ref-36">36</xref>,<xref ref-type="bibr" rid="ref-83">83</xref>]; Yang et al. [<xref ref-type="bibr" rid="ref-36">36</xref>] provide a comprehensive survey of how LLMs are being integrated into text-centric multimodal sentiment analysis pipelines. Finally, <bold>implicit sentiment analysis</bold> addresses cases where sentiment is conveyed indirectly through implications, metaphors, or pragmatic inference rather than explicit opinion words [<xref ref-type="bibr" rid="ref-54">54</xref>,<xref ref-type="bibr" rid="ref-55">55</xref>], a task that has gained renewed attention through chain-of-thought prompting approaches.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Domain Taxonomy</title>
<p>While movie reviews served as the canonical testbed for sentiment analysis [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-41">41</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>,<xref ref-type="bibr" rid="ref-84">84</xref>,<xref ref-type="bibr" rid="ref-85">85</xref>], the field now spans diverse domains with distinct linguistic characteristics. Product reviews on platforms such as Amazon and Yelp feature aspect-heavy text with domain-specific vocabulary [<xref ref-type="bibr" rid="ref-77">77</xref>]. Social media platforms including Twitter/X and Reddit present challenges of informal language, hashtags, emojis, sarcasm, and character limits [<xref ref-type="bibr" rid="ref-8">8</xref>,<xref ref-type="bibr" rid="ref-82">82</xref>,<xref ref-type="bibr" rid="ref-86">86</xref>]. Financial text, including earnings calls, news articles, and analyst reports, requires domain-adapted models; Shen and Zhang [<xref ref-type="bibr" rid="ref-52">52</xref>] reported that GPT-4-class models using few-shot prompting can approach the performance of fine-tuned financial sentiment models such as FinBERT. Bhatia et al. [<xref ref-type="bibr" rid="ref-87">87</xref>] later introduced FinTral, a domain-adapted financial language model based on the Mistral architecture that achieved competitive performance on financial sentiment benchmarks. Healthcare text, encompassing drug reviews, clinical notes, and patient feedback, demands high accuracy and interpretability given the stakes involved. Cross-lingual settings require transferring SA capabilities across languages, particularly to low-resource languages where labeled data is scarce [<xref ref-type="bibr" rid="ref-31">31</xref>,<xref ref-type="bibr" rid="ref-32">32</xref>,<xref ref-type="bibr" rid="ref-61">61</xref>&#x2013;<xref ref-type="bibr" rid="ref-64">64</xref>]. A closely related and practically important application domain is <bold>online hate detection and toxic content analysis</bold>, which shares significant methodological overlap with sentiment analysis while introducing distinct challenges. Hate speech detection requires distinguishing between negative sentiment (legitimate criticism) and harmful content (targeted abuse), often in the presence of implicit toxicity, code-switching, and platform-specific linguistic norms. The TweetEval benchmark [<xref ref-type="bibr" rid="ref-82">82</xref>] includes offensive language detection as one of its unified tasks, and Ranasinghe and Zampieri [<xref ref-type="bibr" rid="ref-88">88</xref>] demonstrated that cross-lingual embeddings can transfer offensive language identification capabilities across languages, an approach directly applicable to multilingual sentiment analysis. The intersection of sentiment analysis with content moderation represents a high-stakes practical frontier where classification errors carry significant social consequences.</p>
<p>The diversity of task formulations, granularity levels, and domains outlined above directly shapes how sentiment analysis systems are evaluated. Models optimized for document-level polarity may perform poorly on aspect-based or implicit sentiment tasks, and benchmarks designed for one domain often fail to reflect challenges present in others. Consequently, the choice of datasets and evaluation metrics plays a decisive role in interpreting reported performance. The following section surveys the benchmark datasets and evaluation protocols that have driven progress in sentiment analysis research.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Benchmark Datasets and Evaluation</title>
<p>Benchmark datasets and evaluation protocols play a central role in shaping progress in sentiment analysis. Beyond serving as performance indicators, benchmarks implicitly define task boundaries, influence model design choices, and determine which linguistic phenomena are emphasized or overlooked. As sentiment analysis has evolved from document-level polarity classification to fine-grained, multilingual, and multimodal tasks, benchmark datasets have likewise diversified in scope, annotation strategies, and evaluation metrics. This section surveys the most widely used benchmark datasets and evaluation methodologies, highlighting how they reflect underlying task formulations and expose both the strengths and limitations of existing sentiment analysis models. <xref ref-type="fig" rid="fig-4">Fig. 4</xref> situates widely used sentiment analysis benchmarks within a historical timeline, highlighting shifts in task formulation and dataset scale.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Timeline of major benchmark datasets in sentiment analysis, illustrating the progression from early document-level datasets to aspect-based, large-scale industrial, and multimodal sentiment benchmarks (This figure was generated with the assistance of OpenAI GPT 5.2).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-4.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Text Benchmarks</title>
<p>The development of standardized benchmarks has been instrumental in driving progress in sentiment analysis, and <xref ref-type="table" rid="table-1">Table 1</xref> summarizes the major text-based datasets that have shaped the field. <bold>The Movie Review (MR)</bold> Corpus, introduced by Pang and Lee [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-89">89</xref>], contains 2000 movie reviews (later expanded to 10,662) labeled as positive or negative; despite its age, it remains frequently used due to its balanced design and linguistic richness. The <bold>IMDb Dataset</bold> [<xref ref-type="bibr" rid="ref-85">85</xref>] provides 50,000 movie reviews with binary sentiment labels and has become the de facto standard for evaluating document-level SA, with reviews that are substantially longer than typical benchmark texts, thereby testing models&#x2019; ability to handle extended context.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Summary of major sentiment analysis benchmark datasets. These are widely used benchmarks in English language (cross-lingual research discussed in <xref ref-type="sec" rid="s5_4">Sections 5.4</xref> and <xref ref-type="sec" rid="s6_6">6.6</xref>).</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th>Dataset</th>
<th>Year</th>
<th>Approx. Size</th>
<th>Classes</th>
<th>Domain</th>
<th>Key Feature</th>
</tr>
</thead>
<tbody>
<tr>
<td>MR Corpus [<xref ref-type="bibr" rid="ref-1">1</xref>]</td>
<td>2002</td>
<td>10.6K</td>
<td>2</td>
<td>Movies</td>
<td>Pioneering SA dataset</td>
</tr>
<tr>
<td>IMDb [<xref ref-type="bibr" rid="ref-85">85</xref>]</td>
<td>2011</td>
<td>50K</td>
<td>2</td>
<td>Movies</td>
<td>Long reviews, balanced</td>
</tr>
<tr>
<td>SST-2/5 [<xref ref-type="bibr" rid="ref-84">84</xref>]</td>
<td>2013</td>
<td>11.8K</td>
<td>2/5</td>
<td>Movies</td>
<td>Phrase-level annotations</td>
</tr>
<tr>
<td>Amazon [<xref ref-type="bibr" rid="ref-29">29</xref>]</td>
<td>2007</td>
<td>8K</td>
<td>2</td>
<td>Products</td>
<td>Multi-domain transfer</td>
</tr>
<tr>
<td>SemEval-2014</td>
<td>2014</td>
<td>6.5K</td>
<td>3</td>
<td>Rest./Lap.</td>
<td>Aspect-based SA</td>
</tr>
<tr>
<td>GoEmotions [<xref ref-type="bibr" rid="ref-81">81</xref>]</td>
<td>2020</td>
<td>58K</td>
<td>28</td>
<td>Reddit</td>
<td>Fine-grained emotion</td>
</tr>
<tr>
<td>TweetEval [<xref ref-type="bibr" rid="ref-82">82</xref>]</td>
<td>2020</td>
<td>124K</td>
<td>Varies</td>
<td>Twitter</td>
<td>Unified tweet tasks</td>
</tr>
<tr>
<td>Fin. PhraseBank</td>
<td>2014</td>
<td>4.8K</td>
<td>3</td>
<td>Finance</td>
<td>Annotator agreement</td>
</tr>
<tr>
<td>SentiEval [<xref ref-type="bibr" rid="ref-22">22</xref>]</td>
<td>2024</td>
<td>26 sets</td>
<td>Varies</td>
<td>Multi</td>
<td>LLM evaluation suite</td>
</tr>
<tr>
<td>CMU-MOSI [<xref ref-type="bibr" rid="ref-90">90</xref>]</td>
<td>2016</td>
<td>2.2K</td>
<td>7</td>
<td>Videos</td>
<td>Multimodal (T&#x002B;A&#x002B;V)</td>
</tr>
<tr>
<td>CMU-MOSEI [<xref ref-type="bibr" rid="ref-83">83</xref>]</td>
<td>2018</td>
<td>23.4K</td>
<td>7</td>
<td>Videos</td>
<td>Large-scale multimodal</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The <bold>Stanford Sentiment Treebank (SST)</bold> [<xref ref-type="bibr" rid="ref-84">84</xref>] represents a landmark contribution: Socher et al. annotated parse trees of 11,855 sentences at every node, enabling both binary (SST-2) and fine-grained five-class (SST-5) evaluation. SST-2 has become one of the most widely reported benchmarks, with BERT achieving 94.9% accuracy [<xref ref-type="bibr" rid="ref-16">16</xref>]. The <bold>SemEval Series</bold> of workshops has produced a rich collection of SA benchmarks, including aspect-based SA datasets (SemEval-2014 Task 4) [<xref ref-type="bibr" rid="ref-27">27</xref>], stance detection (SemEval-2016 Task 6), and figurative language sentiment [<xref ref-type="bibr" rid="ref-27">27</xref>]. More recently, Demszky et al. [<xref ref-type="bibr" rid="ref-81">81</xref>] introduced <bold>GoEmotions</bold>, comprising 58,009 Reddit comments annotated with 27 emotion categories plus neutral, which enables fine-grained emotion detection beyond simple polarity. Barbieri et al. [<xref ref-type="bibr" rid="ref-82">82</xref>] unified seven tweet classification tasks, including sentiment, emotion, and offensive language detection, into a single <bold>TweetEval</bold> benchmark.</p>
<p>While the historical prominence of movie review datasets (MR, IMDb, SST) reflects the field&#x2019;s origins, practitioners should note that the linguistic characteristics of movie reviews: lengthy, grammatically correct, opinion-focused differ substantially from the short, noisy, informal text encountered in social media, financial, and healthcare domains. This discrepancy motivates the continued development of domain-specific benchmarks discussed below.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Multimodal Benchmarks</title>
<p>Multimodal sentiment analysis requires datasets that align textual, acoustic, and visual signals [<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-91">91</xref>]. <bold>CMU-MOSI</bold> [<xref ref-type="bibr" rid="ref-90">90</xref>] contains 2199 video segments from YouTube opinion videos annotated with sentiment intensity on a <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:mo stretchy="false">[</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mo>+</mml:mo><mml:mn>3</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> scale, while <bold>CMU-MOSEI</bold> [<xref ref-type="bibr" rid="ref-83">83</xref>] provides a substantially larger collection of 23,453 annotated segments across 1000 speakers and 250 topics, with annotations for both sentiment and six basic emotions. The <bold>Multimodal EmotionLines Dataset (MELD)</bold> [<xref ref-type="bibr" rid="ref-92">92</xref>] offers 13,708 utterances from the television series <italic>Friends</italic>, annotated with seven emotion labels and three sentiment labels, thereby enabling research on conversational emotion recognition. A recent systematic review [<xref ref-type="bibr" rid="ref-60">60</xref>] analyzing 116 multimodal SA studies from 2018&#x2013;2025 identifies persistent challenges in temporal alignment between modalities and the continued dominance of text over audio-visual features in most fusion architectures.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Domain-Specific Benchmarks</title>
<p>Several benchmarks target specific domains or evaluation objectives. Zhang et al. introduced <bold>SentiEval</bold> [<xref ref-type="bibr" rid="ref-22">22</xref>], a comprehensive benchmark encompassing 13 SA tasks across 26 datasets, specifically designed to evaluate LLM capabilities. Their evaluation revealed that fine-tuned small language models (SLMs) outperform LLMs on most standard tasks, though LLMs demonstrate superior few-shot learning. The <bold>Financial PhraseBank</bold> contains 4845 sentences from financial news annotated by domain experts with three-class sentiment labels; Shen and Zhang [<xref ref-type="bibr" rid="ref-52">52</xref>] used this benchmark to demonstrate that GPT-4o with few-shot prompting achieves performance comparable to fine-tuned FinBERT. In the area of figurative language, Zhang et al. introduced <bold>SarcasmBench</bold> [<xref ref-type="bibr" rid="ref-28">28</xref>], the first comprehensive sarcasm detection benchmark for evaluating LLMs, which revealed that GPT-4 achieves 14% higher accuracy than other LLMs but that all LLMs underperform fine-tuned pre-trained language models (PLMs). <xref ref-type="fig" rid="fig-5">Fig. 5</xref> illustrates the diminishing performance gains observed as sentiment analysis datasets scale to increasingly large sizes.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Illustrative relationship between training dataset size and relative sentiment analysis performance (%). The curve highlights diminishing returns as dataset scale increases, reflecting saturation trends reported across multiple benchmarks rather than results from a single experimental setup (This figure was generated with the assistance of OpenAI GPT 5.2).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-5.tif"/>
</fig>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Evaluation Metrics and Their Limitations</title>
<p>Standard evaluation metrics for sentiment analysis include accuracy, precision, recall, F1-score (macro and weighted), and, for regression tasks, mean absolute error (MAE) and Pearson correlation. For binary classification, accuracy is defined as:</p>
<p><disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mtext>Accuracy</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <italic>TP</italic> (true positives) and <italic>TN</italic> (true negatives) denote the number of correctly classified positive and negative instances, respectively, while <italic>FP</italic> (false positives) and <italic>FN</italic> (false negatives) represent misclassified instances. These quantities are defined with respect to a given positive class in binary classification settings. It is worth mentioning that in multi-class settings, accuracy is computed analogously as the proportion of correctly classified instances over the total number of samples.</p>
<p>Precision and recall provide complementary class-specific information:<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mtext>Precision</mml:mtext></mml:mrow><mml:mi>c</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:msub><mml:mrow><mml:mtext>Recall</mml:mtext></mml:mrow><mml:mi>c</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:mi>T</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mi>F</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula>, and <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:mi>F</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula> denote the true positives, false positives, and false negatives for class <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mi>c</mml:mi></mml:math></inline-formula>, respectively, computed under a one-vs-rest formulation in multi-class classification. <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mi>c</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow></mml:math></inline-formula> corresponds to a sentiment class such as positive, negative, or neutral.</p>
<p>In the SA context, the practical interpretation of these metrics varies by application: high precision for negative sentiment is critical in brand monitoring (minimizing false alarms), while high recall for negative sentiment matters more in mental health surveillance (ensuring no distress signal is missed).</p>
<p>To obtain a single performance measure across all classes, precision and recall are typically aggregated over classes using averaging strategies such as macro-averaging or weighted averaging. The harmonic mean of precision and recall yields the F1-score:<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mtext>F1</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>macro</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:munder><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi>&#x1D4B4;</mml:mi></mml:mrow></mml:mrow></mml:munder><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:msub><mml:mi>P</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:msub><mml:mi>R</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula> denote the precision and recall for class <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:mi>c</mml:mi></mml:math></inline-formula> as defined in <xref ref-type="disp-formula" rid="eqn-4">Eq. (4)</xref>. Macro-averaging assigns equal weight to each class regardless of class frequency, making it particularly suitable for imbalanced datasets where minority classes are of interest. Alternative aggregation strategies include micro-averaging, which aggregates counts globally across all classes, and weighted averaging, which weights each class by its support.</p>
<p>These metrics, however, have well-known limitations in the SA context. Accuracy is misleading on the imbalanced datasets commonly encountered in real-world applications [<xref ref-type="bibr" rid="ref-40">40</xref>], and standard metrics fail to capture the severity of misclassification, confusing positive with negative is arguably worse than confusing positive with neutral. As a result, modern transformer models often achieve very high and tightly clustered accuracy scores, which limits the discriminative power of SST-2 as a benchmark and motivating next-generation evaluations such as SentiEval [<xref ref-type="bibr" rid="ref-22">22</xref>] that test across diverse task types. Recent work by Herrera-Poyatos et al. [<xref ref-type="bibr" rid="ref-25">25</xref>] highlights an additional concern: the model variability problem (MVP), where LLMs produce different sentiment classifications for identical inputs across runs due to stochastic decoding. This phenomenon challenges the reproducibility of reported benchmark scores and demands new evaluation protocols that account for output variance.</p>
<p>A fundamental limitation specific to sentiment analysis is that ground truth itself is subjective: reasonable annotators may disagree on the sentiment of ambiguous or nuanced text. Inter-annotator agreement metrics such as Krippendorff&#x2019;s <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> and Cohen&#x2019;s <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:mi>&#x03BA;</mml:mi></mml:math></inline-formula> therefore serve as a soft upper bound on meaningful model performance. Datasets with low inter-annotator agreement (common in fine-grained and implicit sentiment tasks) constrain the maximum achievable accuracy, making it essential to report agreement statistics alongside model performance. For regression tasks such as multimodal sentiment intensity prediction, Pearson correlation and MAE should be reported jointly, as correlation captures ranking quality while MAE captures calibration. Additionally, the distinction between ordinal and nominal misclassification is often overlooked: standard F1 treats all errors equally, but in ordinal settings (e.g., SST-5), predicting 1-star for a 5-star review is qualitatively worse than predicting 4-star, suggesting that metrics such as mean absolute error or quadratic weighted kappa may be more appropriate.</p>
<p>Taken together, these benchmarks and evaluation protocols have not merely measured progress in sentiment analysis, but actively shaped it. Limitations in early datasets motivated feature-based learning, benchmark saturation exposed the ceilings of classical models, and increasingly complex tasks&#x2014;such as aspect-based, multimodal, and cross-lingual sentiment analysis&#x2014;drove the adoption of more expressive architectures. With this empirical context established, we now trace the methodological evolution of sentiment analysis models, highlighting how successive paradigms emerged in response to these evaluation challenges.</p>
<p>The evolution of benchmark datasets has not only measured progress in sentiment analysis but also shaped it. Early datasets such as MR and IMDb emphasized document-level polarity, enabling steady improvements in classification accuracy, while more recent benchmarks introduce fine-grained, multimodal, and cross-lingual challenges that remain far from solved. Notably, the saturation of performance on widely used datasets such as SST-2 limits their discriminative power, making it increasingly difficult to distinguish between modern architectures. At the same time, the subjective nature of sentiment annotation imposes an inherent ceiling on achievable performance, suggesting that future evaluation frameworks must move beyond static accuracy metrics toward robustness, variability, and real-world deployment criteria.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Methodological Evolution</title>
<p>This section traces the development of sentiment analysis methods across five major paradigms, progressing from rule-based lexicons to trillion-parameter language models. <xref ref-type="table" rid="table-2">Table 2</xref> provides a high-level comparative overview, while <xref ref-type="table" rid="table-3">Table 3</xref> later in this section offers a more granular performance comparison across representative models. Throughout this section, we connect each paradigm back to the formal taxonomy of <xref ref-type="sec" rid="s2">Section 2</xref>, noting how different parameterizations of <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo>;</mml:mo><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> address different granularity levels, task formulations, and domains with varying degrees of success.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Comparative overview of sentiment analysis paradigms.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th>Paradigm</th>
<th>Era</th>
<th>Repr. Model</th>
<th>IMDb Acc.</th>
<th>SST-2 Acc.</th>
<th>Params</th>
<th>Training</th>
</tr>
</thead>
<tbody>
<tr>
<td>Lexicon</td>
<td>2002&#x002B;</td>
<td>VADER</td>
<td><inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>65%&#x2013;70%</td>
<td><inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>60%&#x2013;70%</td>
<td>0</td>
<td>None</td>
</tr>
<tr>
<td>Trad. ML</td>
<td>2002&#x002B;</td>
<td>SVM</td>
<td><inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>86%&#x2013;89%</td>
<td><inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>80%&#x2013;83%</td>
<td><inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>10K</td>
<td>Supervised</td>
</tr>
<tr>
<td>Deep Learn.</td>
<td>2014&#x002B;</td>
<td>BiLSTM-Att</td>
<td><inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>89%&#x2013;92%</td>
<td><inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>87%&#x2013;89%</td>
<td><inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>5M</td>
<td>Supervised</td>
</tr>
<tr>
<td>Transformer</td>
<td>2018&#x002B;</td>
<td>RoBERTa</td>
<td><inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>95%&#x2013;96%</td>
<td>96.4%</td>
<td>355M</td>
<td>Pretrain&#x002B;FT</td>
</tr>
<tr>
<td>LLM</td>
<td>2023&#x002B;</td>
<td>GPT-4</td>
<td><inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>93%&#x2013;96%</td>
<td><inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>95%&#x2013;97%</td>
<td>&#x003E;1T</td>
<td>Zero/Few-shot</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Detailed performance comparison of representative models across paradigms on standard SA benchmarks. Reported accuracies (%) are from original papers or SentiEval [<xref ref-type="bibr" rid="ref-22">22</xref>].</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th>Model</th>
<th>Year</th>
<th>SST-2</th>
<th>IMDb</th>
<th>Params</th>
<th>Paradigm</th>
</tr>
</thead>
<tbody>
<tr>
<td>VADER [<xref ref-type="bibr" rid="ref-8">8</xref>]</td>
<td>2014</td>
<td>65&#x2013;70</td>
<td>65&#x2013;70</td>
<td>&#x2013;</td>
<td>Lexicon</td>
</tr>
<tr>
<td>SVM&#x002B;BoW [<xref ref-type="bibr" rid="ref-1">1</xref>]</td>
<td>2002</td>
<td>78&#x2013;82</td>
<td>86&#x2013;89</td>
<td><inline-formula id="ieqn-50"><mml:math id="mml-ieqn-50"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>10K</td>
<td>Trad. ML</td>
</tr>
<tr>
<td>TextCNN [<xref ref-type="bibr" rid="ref-12">12</xref>]</td>
<td>2014</td>
<td>87&#x2013;88</td>
<td>90&#x2013;91</td>
<td><inline-formula id="ieqn-51"><mml:math id="mml-ieqn-51"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>1M</td>
<td>Deep Learn.</td>
</tr>
<tr>
<td>BiLSTM-Att [<xref ref-type="bibr" rid="ref-15">15</xref>]</td>
<td>2016</td>
<td>88&#x2013;90</td>
<td>90&#x2013;92</td>
<td><inline-formula id="ieqn-52"><mml:math id="mml-ieqn-52"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>5M</td>
<td>Deep Learn.</td>
</tr>
<tr>
<td>BERT-base [<xref ref-type="bibr" rid="ref-16">16</xref>]</td>
<td>2018</td>
<td>94.9</td>
<td>93&#x2013;95</td>
<td>110M</td>
<td>Transformer (FT)</td>
</tr>
<tr>
<td>RoBERTa [<xref ref-type="bibr" rid="ref-17">17</xref>]</td>
<td>2019</td>
<td>96.4</td>
<td>95&#x2013;96</td>
<td>355M</td>
<td>Transformer (FT)</td>
</tr>
<tr>
<td>XLNet [<xref ref-type="bibr" rid="ref-18">18</xref>]</td>
<td>2019</td>
<td>96&#x2013;97</td>
<td>95&#x2013;96</td>
<td>340M</td>
<td>Transformer</td>
</tr>
<tr>
<td>ModernBERT [<xref ref-type="bibr" rid="ref-58">58</xref>]</td>
<td>2024</td>
<td>96&#x2013;97</td>
<td>95&#x2013;96</td>
<td>395M</td>
<td>Transformer (FT)</td>
</tr>
<tr>
<td>GPT-3.5 (0-shot) [<xref ref-type="bibr" rid="ref-22">22</xref>]</td>
<td>2023</td>
<td>88&#x2013;92</td>
<td>88&#x2013;91</td>
<td>175B</td>
<td>LLM (Zero-shot)</td>
</tr>
<tr>
<td>GPT-4 (0-shot) [<xref ref-type="bibr" rid="ref-22">22</xref>]</td>
<td>2024</td>
<td>94&#x2013;96</td>
<td>93&#x2013;95</td>
<td><inline-formula id="ieqn-53"><mml:math id="mml-ieqn-53"><mml:msup><mml:mi>&#x003E;1T</mml:mi><mml:mo>&#x2020;</mml:mo></mml:msup></mml:math></inline-formula></td>
<td>LLM (Zero-shot)</td>
</tr>
<tr>
<td>GPT-4 (5-shot) [<xref ref-type="bibr" rid="ref-22">22</xref>]</td>
<td>2024</td>
<td>95&#x2013;97</td>
<td>94&#x2013;96</td>
<td><inline-formula id="ieqn-54"><mml:math id="mml-ieqn-54"><mml:msup><mml:mi>&#x003E;1T</mml:mi><mml:mo>&#x2020;</mml:mo></mml:msup></mml:math></inline-formula></td>
<td>LLM (Few-shot)</td>
</tr>
<tr>
<td>LLaMA-3-70B (0-shot)</td>
<td>2024</td>
<td>93&#x2013;95</td>
<td>92&#x2013;94</td>
<td>70B</td>
<td>LLM (Zero-shot)</td>
</tr>
<tr>
<td>Mixtral-8x22B (0-shot)</td>
<td>2024</td>
<td>92&#x2013;95</td>
<td>92&#x2013;94</td>
<td>141B&#x002A;</td>
<td>MoE LLM (Zero-shot)</td>
</tr>
<tr>
<td>Gemini 1.5 Pro (0-shot)</td>
<td>2025</td>
<td>94&#x2013;96</td>
<td>93&#x2013;95</td>
<td><inline-formula id="ieqn-55"><mml:math id="mml-ieqn-55"><mml:msup><mml:mi>&#x003E;500B</mml:mi><mml:mo>&#x2020;</mml:mo></mml:msup></mml:math></inline-formula></td>
<td>LLM (Zero-shot)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-3fn1" fn-type="other">
<p><inline-formula id="ieqn-56"><mml:math id="mml-ieqn-56"><mml:msup><mml:mi>Note:</mml:mi><mml:mrow><mml:mo>&#x2217;</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula>Total parameters in MoE; active parameters per token are lower. <sup>&#x2020;</sup>Estimated model scale based on publicly available information. Recent LLM results are based on prompt-based zero-shot evaluations and are not directly comparable to fine-tuned models.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<sec id="s4_1">
<label>4.1</label>
<title>Lexicon-Based Methods</title>
<p>Lexicon-based approaches represent the earliest systematic methods for sentiment analysis, relying on pre-compiled dictionaries that associate words or phrases with sentiment scores [<xref ref-type="bibr" rid="ref-6">6</xref>&#x2013;<xref ref-type="bibr" rid="ref-8">8</xref>,<xref ref-type="bibr" rid="ref-72">72</xref>,<xref ref-type="bibr" rid="ref-93">93</xref>,<xref ref-type="bibr" rid="ref-94">94</xref>]. SentiWordNet [<xref ref-type="bibr" rid="ref-6">6</xref>,<xref ref-type="bibr" rid="ref-7">7</xref>] assigns positivity, negativity, and objectivity scores to each WordNet synset, enabling the computation of aggregate document sentiment as:<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:munder><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>w</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>(</mml:mo><mml:msup><mml:mi>s</mml:mi><mml:mo>+</mml:mo></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>w</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msup><mml:mi>s</mml:mi><mml:mo>&#x2212;</mml:mo></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>w</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-58"><mml:math id="mml-ieqn-58"><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:math></inline-formula> denotes the number of tokens in document <inline-formula id="ieqn-59"><mml:math id="mml-ieqn-59"><mml:mi>d</mml:mi></mml:math></inline-formula>, and <inline-formula id="ieqn-60"><mml:math id="mml-ieqn-60"><mml:msup><mml:mi>s</mml:mi><mml:mo>+</mml:mo></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>w</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> and <inline-formula id="ieqn-61"><mml:math id="mml-ieqn-61"><mml:msup><mml:mi>s</mml:mi><mml:mo>&#x2212;</mml:mo></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>w</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represent the positive and negative sentiment scores assigned to word <inline-formula id="ieqn-62"><mml:math id="mml-ieqn-62"><mml:mi>w</mml:mi></mml:math></inline-formula> by the underlying lexicon. This formulation assumes that each word contributes independently and equally to the overall sentiment, ignoring contextual effects such as negation, intensification, and compositional semantics. The resulting score <inline-formula id="ieqn-63"><mml:math id="mml-ieqn-63"><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is an aggregate polarity measure whose range depends on the underlying lexicon scores and document length.</p>
<p>VADER (Valence Aware Dictionary and Sentiment Reasoner) [<xref ref-type="bibr" rid="ref-8">8</xref>] extended this approach with rule-based heuristics for handling negation, intensifiers, and punctuation, computing a composite score through:<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mrow><mml:mtext>compound</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mi>v</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:msqrt><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mi>v</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:mi>&#x03B1;</mml:mi></mml:msqrt></mml:mfrac></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-64"><mml:math id="mml-ieqn-64"><mml:mi>n</mml:mi></mml:math></inline-formula> denotes the number of tokens in the input text for which valence scores are computed, and <inline-formula id="ieqn-65"><mml:math id="mml-ieqn-65"><mml:msub><mml:mi>v</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> represents the valence score of the <inline-formula id="ieqn-66"><mml:math id="mml-ieqn-66"><mml:mi>i</mml:mi></mml:math></inline-formula>-th token after applying rule-based adjustments for factors such as negation, degree modifiers, punctuation, and capitalization. The parameter <inline-formula id="ieqn-67"><mml:math id="mml-ieqn-67"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> is a normalization constant (typically set to 15) that controls the scaling of the score and prevents extreme values for large aggregated valence magnitudes. This normalization maps the aggregated valence score to a bounded range in <inline-formula id="ieqn-68"><mml:math id="mml-ieqn-68"><mml:mo stretchy="false">[</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula>, facilitating consistent comparison across texts of varying lengths.</p>
<p>A third notable tool, LIWC (Linguistic Inquiry and Word Count) [<xref ref-type="bibr" rid="ref-93">93</xref>,<xref ref-type="bibr" rid="ref-94">94</xref>], provides psychologically validated word categories that extend beyond simple polarity to capture cognitive processes, social dynamics, and affective dimensions, and has been widely applied to sentiment analysis in social media and healthcare contexts despite being primarily designed for psychological research.</p>
<p>The strengths of lexicon-based methods lie in their interpretability, domain independence, and zero-resource applicability. However, they fundamentally cannot handle context-dependent sentiment, compositional semantics (e.g., &#x201C;not bad&#x201D;), sarcasm, or domain-specific terminology [<xref ref-type="bibr" rid="ref-4">4</xref>,<xref ref-type="bibr" rid="ref-95">95</xref>]. Kennedy and Inkpen [<xref ref-type="bibr" rid="ref-95">95</xref>] demonstrated that contextual valence shifters can partially address negation, but systematic limitations remain, motivating the transition to data-driven approaches.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Traditional Machine Learning</title>
<p>Machine learning approaches reformulated sentiment analysis as a supervised classification problem, learning discriminative patterns from labeled data [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-9">9</xref>&#x2013;<xref ref-type="bibr" rid="ref-11">11</xref>].</p>
<sec id="s4_2_1">
<label>4.2.1</label>
<title>Support Vector Machines</title>
<p>SVMs became the dominant approach for text classification following Joachims&#x2019; seminal work [<xref ref-type="bibr" rid="ref-9">9</xref>]. For sentiment analysis, each document is represented as a feature vector <inline-formula id="ieqn-69"><mml:math id="mml-ieqn-69"><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mi>d</mml:mi></mml:msup></mml:math></inline-formula> (typically using bag-of-words or TF-IDF features) and classified by finding the maximum-margin hyperplane. The SVM optimization problem takes the form:<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi></mml:mi><mml:munder><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo>,</mml:mo><mml:mi mathvariant="bold-italic">&#x03BE;</mml:mi></mml:mrow></mml:munder><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:mi>C</mml:mi><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mi>&#x03BE;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mspace width="1em" /><mml:mrow><mml:mtext>s.t.</mml:mtext></mml:mrow><mml:mspace width="1em" /><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2265;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>&#x03BE;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>&#x03BE;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2265;</mml:mo><mml:mn>0</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-70"><mml:math id="mml-ieqn-70"><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow></mml:math></inline-formula> is the weight vector defining the separating hyperplane, <inline-formula id="ieqn-71"><mml:math id="mml-ieqn-71"><mml:mi>b</mml:mi></mml:math></inline-formula> is the bias term, and <inline-formula id="ieqn-72"><mml:math id="mml-ieqn-72"><mml:msub><mml:mi>&#x03BE;</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> are slack variables that allow for margin violations in non-separable data. The parameter <inline-formula id="ieqn-73"><mml:math id="mml-ieqn-73"><mml:mi>C</mml:mi><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> controls the trade-off between maximizing the margin (via minimizing <inline-formula id="ieqn-74"><mml:math id="mml-ieqn-74"><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:msup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula>) and penalizing classification errors through the slack variables. Here, <inline-formula id="ieqn-75"><mml:math id="mml-ieqn-75"><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula> denotes the binary class label associated with input <inline-formula id="ieqn-76"><mml:math id="mml-ieqn-76"><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula>. This formulation corresponds to the soft-margin support vector machine, which allows for misclassification in exchange for improved generalization on non-linearly separable data.</p>
<p>The dual formulation with kernel function <inline-formula id="ieqn-77"><mml:math id="mml-ieqn-77"><mml:mi>K</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> enables non-linear classification:<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mtext>sgn</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mi>K</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>b</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-78"><mml:math id="mml-ieqn-78"><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> are the Lagrange multipliers obtained from the dual optimization problem in <xref ref-type="disp-formula" rid="eqn-8">Eq. (8)</xref>, and <inline-formula id="ieqn-79"><mml:math id="mml-ieqn-79"><mml:mi>K</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> denotes a kernel function that computes an inner product in a (possibly high-dimensional) feature space, enabling non-linear classification. Only a subset of training instances with <inline-formula id="ieqn-80"><mml:math id="mml-ieqn-80"><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> (the support vectors) contribute to the decision function. The <inline-formula id="ieqn-81"><mml:math id="mml-ieqn-81"><mml:mi>sgn</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> function returns the sign of its argument, determining the predicted class label.</p>
<p>Pang et al. [<xref ref-type="bibr" rid="ref-1">1</xref>] demonstrated that SVMs with unigram features achieved 82.9% accuracy on their movie review corpus, significantly outperforming lexicon-based methods and establishing machine learning as the dominant paradigm. Subsequent work explored n-gram features, part-of-speech tags [<xref ref-type="bibr" rid="ref-96">96</xref>], and domain adaptation techniques [<xref ref-type="bibr" rid="ref-29">29</xref>].</p>
</sec>
<sec id="s4_2_2">
<label>4.2.2</label>
<title>Na&#x00EF;ve Bayes</title>
<p>The Na&#x00EF;ve Bayes classifier assumes feature independence and applies Bayes&#x2019; theorem [<xref ref-type="bibr" rid="ref-10">10</xref>,<xref ref-type="bibr" rid="ref-97">97</xref>]:<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:munderover><mml:mo>&#x220F;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>Here, <inline-formula id="ieqn-82"><mml:math id="mml-ieqn-82"><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represents the feature vector corresponding to the input text, where each <inline-formula id="ieqn-83"><mml:math id="mml-ieqn-83"><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> typically denotes the presence or frequency of a token. This formulation relies on the Na&#x00EF;ve Bayes assumption that features are conditionally independent given the class label, i.e., <inline-formula id="ieqn-84"><mml:math id="mml-ieqn-84"><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo>&#x2223;</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:msubsup><mml:mo movablelimits="false">&#x220F;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>. The denominator <inline-formula id="ieqn-85"><mml:math id="mml-ieqn-85"><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> acts as a normalization constant and is typically omitted during prediction, as it is independent of the class label. In practice, classification is performed by selecting the class with the highest posterior probability, i.e., <inline-formula id="ieqn-86"><mml:math id="mml-ieqn-86"><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mi>arg</mml:mi><mml:mo>&#x2061;</mml:mo><mml:munder><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mi>y</mml:mi></mml:munder><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>.</p>
<p>Despite the strong independence assumption, Na&#x00EF;ve Bayes performs surprisingly well for text classification due to the high dimensionality of the feature space, achieving competitive results with far less training time than SVMs [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-97">97</xref>]. The multinomial variant, which models word counts rather than binary presence, is particularly effective for longer documents [<xref ref-type="bibr" rid="ref-10">10</xref>].</p>
</sec>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Deep Learning Approaches</title>
<p>Deep learning methods introduced the ability to learn hierarchical feature representations directly from raw text, eliminating the need for manual feature engineering [<xref ref-type="bibr" rid="ref-2">2</xref>,<xref ref-type="bibr" rid="ref-98">98</xref>,<xref ref-type="bibr" rid="ref-99">99</xref>]. This subsection traces the key architectural developments that shaped the deep learning era of sentiment analysis.</p>
<sec id="s4_3_1">
<label>4.3.1</label>
<title>Word Embeddings</title>
<p>The transition from sparse bag-of-words representations to dense, continuous word embeddings marked a foundational shift. Word2Vec [<xref ref-type="bibr" rid="ref-100">100</xref>,<xref ref-type="bibr" rid="ref-101">101</xref>] learns distributed representations through either the skip-gram or continuous bag-of-words (CBOW) objective. The skip-gram model maximizes:<disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>T</mml:mi></mml:mfrac><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover><mml:munder><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>c</mml:mi><mml:mo>&#x2264;</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x2264;</mml:mo><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x2260;</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:munder><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <italic>T</italic> denotes the total number of tokens in the corpus, <inline-formula id="ieqn-87"><mml:math id="mml-ieqn-87"><mml:msub><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> is the token at position <inline-formula id="ieqn-88"><mml:math id="mml-ieqn-88"><mml:mi>t</mml:mi></mml:math></inline-formula>, and <inline-formula id="ieqn-89"><mml:math id="mml-ieqn-89"><mml:mi>c</mml:mi></mml:math></inline-formula> is the context window size defining how many neighboring tokens are considered. In practice, the summation over context positions is restricted to valid indices within the sequence boundaries. The conditional probability <inline-formula id="ieqn-90"><mml:math id="mml-ieqn-90"><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is typically parameterized using a softmax function over the vocabulary based on learned word embeddings. This objective maximizes the likelihood of observing context words given a center word, thereby learning distributed word representations that capture semantic relationships.</p>
<p>GloVe [<xref ref-type="bibr" rid="ref-102">102</xref>] subsequently combined global matrix factorization with local context windows by optimizing a weighted least-squares objective on co-occurrence statistics:<disp-formula id="eqn-12"><label>(12)</label><mml:math id="mml-eqn-12" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>J</mml:mi><mml:mo>=</mml:mo><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:munderover><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mi>i</mml:mi><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msubsup><mml:msub><mml:mrow><mml:mover><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>j</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>b</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>j</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-91"><mml:math id="mml-ieqn-91"><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the co-occurrence count of word <inline-formula id="ieqn-92"><mml:math id="mml-ieqn-92"><mml:mi>j</mml:mi></mml:math></inline-formula> in the context of word <inline-formula id="ieqn-93"><mml:math id="mml-ieqn-93"><mml:mi>i</mml:mi></mml:math></inline-formula> within a predefined window over the corpus. Here, <inline-formula id="ieqn-94"><mml:math id="mml-ieqn-94"><mml:msub><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-95"><mml:math id="mml-ieqn-95"><mml:msub><mml:mrow><mml:mover><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula> are the word and context embedding vectors for words <inline-formula id="ieqn-96"><mml:math id="mml-ieqn-96"><mml:mi>i</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-97"><mml:math id="mml-ieqn-97"><mml:mi>j</mml:mi></mml:math></inline-formula>, respectively, and <inline-formula id="ieqn-98"><mml:math id="mml-ieqn-98"><mml:msub><mml:mi>b</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-99"><mml:math id="mml-ieqn-99"><mml:msub><mml:mrow><mml:mover><mml:mi>b</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>j</mml:mi></mml:msub></mml:math></inline-formula> are their associated bias terms. The function <inline-formula id="ieqn-100"><mml:math id="mml-ieqn-100"><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is a weighting function that down-weights the influence of extremely frequent or rare co-occurrences to improve training stability. This objective enforces that the dot product of word embeddings approximates the logarithm of word co-occurrence counts, thereby capturing global statistical structure of the corpus. These embeddings capture semantic relationships (e.g., &#x201C;king&#x201D; &#x2212; &#x201C;man&#x201D; &#x002B; &#x201C;woman&#x201D; <inline-formula id="ieqn-101"><mml:math id="mml-ieqn-101"><mml:mo>&#x2248;</mml:mo></mml:math></inline-formula> &#x201C;queen&#x201D;) and provide substantially better input representations for downstream SA models [<xref ref-type="bibr" rid="ref-85">85</xref>,<xref ref-type="bibr" rid="ref-102">102</xref>].</p>
</sec>
<sec id="s4_3_2">
<label>4.3.2</label>
<title>Convolutional Neural Networks</title>
<p>Kim [<xref ref-type="bibr" rid="ref-12">12</xref>] introduced the TextCNN architecture, applying 1D convolutions over word embedding sequences. For an input sentence represented as a matrix <inline-formula id="ieqn-102"><mml:math id="mml-ieqn-102"><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> (where <inline-formula id="ieqn-103"><mml:math id="mml-ieqn-103"><mml:mi>n</mml:mi></mml:math></inline-formula> is the sequence length and <inline-formula id="ieqn-104"><mml:math id="mml-ieqn-104"><mml:mi>d</mml:mi></mml:math></inline-formula> is the embedding dimension), a convolution filter <inline-formula id="ieqn-105"><mml:math id="mml-ieqn-105"><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> of window size <inline-formula id="ieqn-106"><mml:math id="mml-ieqn-106"><mml:mi>h</mml:mi></mml:math></inline-formula> produces a feature map:<disp-formula id="eqn-13"><label>(13)</label><mml:math id="mml-eqn-13" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mtext>ReLU</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>:</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mi>h</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-107"><mml:math id="mml-ieqn-107"><mml:msub><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>:</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mi>h</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> denotes the submatrix of input embeddings corresponding to a window of <inline-formula id="ieqn-108"><mml:math id="mml-ieqn-108"><mml:mi>h</mml:mi></mml:math></inline-formula> consecutive tokens starting at position <inline-formula id="ieqn-109"><mml:math id="mml-ieqn-109"><mml:mi>i</mml:mi></mml:math></inline-formula>. Here, <inline-formula id="ieqn-110"><mml:math id="mml-ieqn-110"><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is a convolutional filter (kernel) and <inline-formula id="ieqn-111"><mml:math id="mml-ieqn-111"><mml:mi>b</mml:mi></mml:math></inline-formula> is a scalar bias term. The operation <inline-formula id="ieqn-112"><mml:math id="mml-ieqn-112"><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>:</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mi>h</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> denotes the element-wise inner product between the filter and the input window, typically implemented as a flattened dot product. The resulting value <inline-formula id="ieqn-113"><mml:math id="mml-ieqn-113"><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> represents a scalar feature capturing the presence of a specific pattern in the input at position <inline-formula id="ieqn-114"><mml:math id="mml-ieqn-114"><mml:mi>i</mml:mi></mml:math></inline-formula>. The <inline-formula id="ieqn-115"><mml:math id="mml-ieqn-115"><mml:mrow><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">L</mml:mi><mml:mi mathvariant="normal">U</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> activation function introduces non-linearity by mapping negative values to zero.</p>
<p>Max-over-time pooling then extracts the most salient feature: <inline-formula id="ieqn-116"><mml:math id="mml-ieqn-116"><mml:mrow><mml:mover><mml:mi>c</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>h</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>. Multiple filters with varying window sizes (typically <inline-formula id="ieqn-117"><mml:math id="mml-ieqn-117"><mml:mi>h</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>4</mml:mn><mml:mo>,</mml:mo><mml:mn>5</mml:mn><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula>) capture n-gram patterns at different scales. Kim&#x2019;s TextCNN achieved 88.1% on SST-2, demonstrating that even relatively simple architectures with pre-trained embeddings could yield strong SA performance [<xref ref-type="bibr" rid="ref-12">12</xref>].</p>
</sec>
<sec id="s4_3_3">
<label>4.3.3</label>
<title>Recurrent Neural Networks and LSTM</title>
<p>Recurrent neural networks (RNNs) process sequences token-by-token, maintaining a hidden state that captures sequential dependencies. However, vanilla RNNs suffer from the vanishing gradient problem [<xref ref-type="bibr" rid="ref-103">103</xref>], which limits their ability to learn long-range dependencies. The Long Short-Term Memory (LSTM) architecture [<xref ref-type="bibr" rid="ref-13">13</xref>] addresses this through a gating mechanism that regulates information flow:<disp-formula id="eqn-14"><label>(14)</label><mml:math id="mml-eqn-14" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="bold">f</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>f</mml:mi></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">]</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">b</mml:mi></mml:mrow><mml:mi>f</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="1em" /><mml:mspace width="1em" /><mml:mspace width="thinmathspace" /><mml:mrow><mml:mtext>(forget gate)</mml:mtext></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-15"><label>(15)</label><mml:math id="mml-eqn-15" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="bold">i</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">]</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">b</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="1em" /><mml:mspace width="1em" /><mml:mspace width="thinmathspace" /><mml:mspace width="thinmathspace" /><mml:mspace width="thinmathspace" /><mml:mrow><mml:mtext>(input gate)</mml:mtext></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-16"><label>(16)</label><mml:math id="mml-eqn-16" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mover><mml:mrow><mml:mi mathvariant="bold">c</mml:mi></mml:mrow><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>tanh</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>c</mml:mi></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">]</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">b</mml:mi></mml:mrow><mml:mi>c</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="1em" /><mml:mrow><mml:mtext>(candidate)</mml:mtext></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-17"><label>(17)</label><mml:math id="mml-eqn-17" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="bold">c</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">f</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x2299;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">c</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">i</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x2299;</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mrow><mml:mi mathvariant="bold">c</mml:mi></mml:mrow><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mrow><mml:mtext>(cell state)</mml:mtext></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-18"><label>(18)</label><mml:math id="mml-eqn-18" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="bold">o</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>o</mml:mi></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">]</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">b</mml:mi></mml:mrow><mml:mi>o</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="1em" /><mml:mrow><mml:mtext>(output gate)</mml:mtext></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-19"><label>(19)</label><mml:math id="mml-eqn-19" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">o</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x2299;</mml:mo><mml:mi>tanh</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">c</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mspace width="2em" /><mml:mrow><mml:mtext>(hidden state)</mml:mtext></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-118"><mml:math id="mml-ieqn-118"><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> is the input vector at time step <inline-formula id="ieqn-119"><mml:math id="mml-ieqn-119"><mml:mi>t</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-120"><mml:math id="mml-ieqn-120"><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is the hidden state from the previous time step, and <inline-formula id="ieqn-121"><mml:math id="mml-ieqn-121"><mml:msub><mml:mrow><mml:mi mathvariant="bold">c</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> denotes the cell state that carries long-term information. The notation <inline-formula id="ieqn-122"><mml:math id="mml-ieqn-122"><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> denotes the concatenation of the previous hidden state and current input vector, which is linearly transformed by weight matrices <inline-formula id="ieqn-123"><mml:math id="mml-ieqn-123"><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>f</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>c</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:math></inline-formula> and <inline-formula id="ieqn-124"><mml:math id="mml-ieqn-124"><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>o</mml:mi></mml:msub></mml:math></inline-formula> with corresponding biases. The sigmoid function <inline-formula id="ieqn-125"><mml:math id="mml-ieqn-125"><mml:mi>&#x03C3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> maps values to the range <inline-formula id="ieqn-126"><mml:math id="mml-ieqn-126"><mml:mo stretchy="false">[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> and is used for gating, while <inline-formula id="ieqn-127"><mml:math id="mml-ieqn-127"><mml:mi>tanh</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> produces values in <inline-formula id="ieqn-128"><mml:math id="mml-ieqn-128"><mml:mo stretchy="false">[</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> and is used to generate candidate cell states. The operator <inline-formula id="ieqn-129"><mml:math id="mml-ieqn-129"><mml:mo>&#x2299;</mml:mo></mml:math></inline-formula> denotes element-wise multiplication. The forget gate <inline-formula id="ieqn-130"><mml:math id="mml-ieqn-130"><mml:msub><mml:mrow><mml:mi mathvariant="bold">f</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> controls which information from the previous cell state is retained, the input gate <inline-formula id="ieqn-131"><mml:math id="mml-ieqn-131"><mml:msub><mml:mrow><mml:mi mathvariant="bold">i</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> determines how much new information is incorporated, and the output gate <inline-formula id="ieqn-132"><mml:math id="mml-ieqn-132"><mml:msub><mml:mrow><mml:mi mathvariant="bold">o</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> regulates the information exposed to the hidden state. The cell state <inline-formula id="ieqn-133"><mml:math id="mml-ieqn-133"><mml:msub><mml:mrow><mml:mi mathvariant="bold">c</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> serves as a memory mechanism that enables the network to preserve information over long time horizons, mitigating the vanishing gradient problem in standard recurrent networks.</p>
<p>Tai et al. [<xref ref-type="bibr" rid="ref-104">104</xref>] extended LSTMs to tree-structured topologies (Tree-LSTM) that operate over parse trees, achieving 51.0% accuracy on the fine-grained SST-5 task. Bidirectional LSTMs (BiLSTMs) process sequences in both directions, and when combined with attention mechanisms [<xref ref-type="bibr" rid="ref-15">15</xref>], they achieved state-of-the-art performance in the pre-transformer era. More recently, Nkhata et al. [<xref ref-type="bibr" rid="ref-105">105</xref>] demonstrated that fine-tuning BERT with a bidirectional LSTM layer produces further improvements for fine-grained movie review sentiment classification, suggesting that recurrent layers can still complement transformer representations. <xref ref-type="fig" rid="fig-6">Fig. 6</xref> illustrates the structural differences between vanilla RNNs and gated recurrent architectures that motivated their adoption in early sentiment analysis models.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Comparison of recurrent neural network architectures. Vanilla RNNs propagate information through a single hidden state, while gated variants such as LSTM and GRU introduce explicit control mechanisms to regulate information flow and mitigate the vanishing gradient problem. These architectures formed the backbone of pre-transformer deep learning approaches to sentiment analysis. Adapted from [<xref ref-type="bibr" rid="ref-106">106</xref>].</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-6.tif"/>
</fig>
</sec>
<sec id="s4_3_4">
<label>4.3.4</label>
<title>Attention Mechanisms</title>
<p>The attention mechanism, introduced by Bahdanau et al. [<xref ref-type="bibr" rid="ref-14">14</xref>], allows models to selectively focus on relevant parts of the input. For sentiment analysis, attention-based models compute a weighted sum of hidden states:<disp-formula id="eqn-20"><label>(20)</label><mml:math id="mml-eqn-20" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:mrow><mml:mi mathvariant="bold">v</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-134"><mml:math id="mml-ieqn-134"><mml:msub><mml:mi>e</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="bold">u</mml:mi></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mi>tanh</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>h</mml:mi></mml:msub><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="bold">b</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is the attention energy. <inline-formula id="ieqn-135"><mml:math id="mml-ieqn-135"><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> denotes the hidden representation of the input at position <inline-formula id="ieqn-136"><mml:math id="mml-ieqn-136"><mml:mi>t</mml:mi></mml:math></inline-formula>, typically obtained from a recurrent or encoder-based model. The scalar <inline-formula id="ieqn-137"><mml:math id="mml-ieqn-137"><mml:msub><mml:mi>e</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> represents the unnormalized relevance score of the <inline-formula id="ieqn-138"><mml:math id="mml-ieqn-138"><mml:mi>t</mml:mi></mml:math></inline-formula>-th input with respect to the attention mechanism, computed using learnable parameters <inline-formula id="ieqn-139"><mml:math id="mml-ieqn-139"><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>h</mml:mi></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-140"><mml:math id="mml-ieqn-140"><mml:mrow><mml:mi mathvariant="bold">u</mml:mi></mml:mrow></mml:math></inline-formula>, and bias <inline-formula id="ieqn-141"><mml:math id="mml-ieqn-141"><mml:mrow><mml:mi mathvariant="bold">b</mml:mi></mml:mrow></mml:math></inline-formula>. The coefficients <inline-formula id="ieqn-142"><mml:math id="mml-ieqn-142"><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> are attention weights obtained via a softmax function, ensuring that <inline-formula id="ieqn-143"><mml:math id="mml-ieqn-143"><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mi>&#x03B1;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>, and quantify the relative importance of each input position. The vector <inline-formula id="ieqn-144"><mml:math id="mml-ieqn-144"><mml:mrow><mml:mi mathvariant="bold">v</mml:mi></mml:mrow></mml:math></inline-formula> is a context vector computed as a weighted sum of hidden representations, aggregating relevant information across the sequence.</p>
<p>Wang et al. [<xref ref-type="bibr" rid="ref-15">15</xref>] applied aspect-specific attention to LSTM hidden states for ABSA, conditioning the attention weights on the aspect embedding. Yang et al. [<xref ref-type="bibr" rid="ref-107">107</xref>] introduced hierarchical attention networks (HAN) with word-level and sentence-level attention for document classification, naturally capturing the multi-granularity structure of sentiment in long documents.</p>
<p>Despite their success, deep learning models based on CNNs and RNNs remained fundamentally limited by their reliance on task-specific training and their inability to fully exploit large-scale unlabeled text. These constraints motivated a shift toward pre-trained language models that could acquire general linguistic knowledge once and transfer it efficiently across sentiment analysis tasks.</p>
</sec>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Pre-Trained Transformers</title>
<p>The transformer architecture [<xref ref-type="bibr" rid="ref-108">108</xref>] and its pre-trained variants fundamentally transformed sentiment analysis by introducing large-scale transfer learning, enabling models pre-trained on massive unlabeled corpora to be adapted to sentiment tasks with relatively small amounts of labeled data.</p>
<sec id="s4_4_1">
<label>4.4.1</label>
<title>The Transformer Architecture</title>
<p>The transformer [<xref ref-type="bibr" rid="ref-108">108</xref>] replaces recurrence with <italic>self-attention</italic>, computing pairwise interactions between all tokens in parallel. Given an input representation matrix <inline-formula id="ieqn-145"><mml:math id="mml-ieqn-145"><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, the self-attention mechanism computes contextualized representations by projecting <inline-formula id="ieqn-146"><mml:math id="mml-ieqn-146"><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math></inline-formula> into query (<inline-formula id="ieqn-147"><mml:math id="mml-ieqn-147"><mml:mrow><mml:mi mathvariant="bold">Q</mml:mi></mml:mrow></mml:math></inline-formula>), key (<inline-formula id="ieqn-148"><mml:math id="mml-ieqn-148"><mml:mrow><mml:mi mathvariant="bold">K</mml:mi></mml:mrow></mml:math></inline-formula>), and value (<inline-formula id="ieqn-149"><mml:math id="mml-ieqn-149"><mml:mrow><mml:mi mathvariant="bold">V</mml:mi></mml:mrow></mml:math></inline-formula>) matrices and applying scaled dot-product attention:<disp-formula id="eqn-21"><label>(21)</label><mml:math id="mml-eqn-21" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mtext>Attention</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">Q</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="bold">K</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="bold">V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mtext>softmax</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mi mathvariant="bold">Q</mml:mi></mml:mrow><mml:msup><mml:mrow><mml:mi mathvariant="bold">K</mml:mi></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup></mml:mrow><mml:msqrt><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:msqrt></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="bold">V</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>In the input representation matrix denoted by <inline-formula id="ieqn-150"><mml:math id="mml-ieqn-150"><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, <inline-formula id="ieqn-151"><mml:math id="mml-ieqn-151"><mml:mi>n</mml:mi></mml:math></inline-formula> is the sequence length and <inline-formula id="ieqn-152"><mml:math id="mml-ieqn-152"><mml:mi>d</mml:mi></mml:math></inline-formula> the embedding dimension. The queries, keys, and values are obtained as learned linear projections of <inline-formula id="ieqn-153"><mml:math id="mml-ieqn-153"><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math></inline-formula>, i.e., <inline-formula id="ieqn-154"><mml:math id="mml-ieqn-154"><mml:mrow><mml:mi mathvariant="bold">Q</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>Q</mml:mi></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-155"><mml:math id="mml-ieqn-155"><mml:mrow><mml:mi mathvariant="bold">K</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>K</mml:mi></mml:msub></mml:math></inline-formula>, and <inline-formula id="ieqn-156"><mml:math id="mml-ieqn-156"><mml:mrow><mml:mi mathvariant="bold">V</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>V</mml:mi></mml:msub></mml:math></inline-formula>, where <inline-formula id="ieqn-157"><mml:math id="mml-ieqn-157"><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>Q</mml:mi></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-158"><mml:math id="mml-ieqn-158"><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>K</mml:mi></mml:msub></mml:math></inline-formula>, and <inline-formula id="ieqn-159"><mml:math id="mml-ieqn-159"><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>V</mml:mi></mml:msub></mml:math></inline-formula> are trainable weight matrices. Typically, <inline-formula id="ieqn-160"><mml:math id="mml-ieqn-160"><mml:mrow><mml:mi mathvariant="bold">Q</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>, <inline-formula id="ieqn-161"><mml:math id="mml-ieqn-161"><mml:mrow><mml:mi mathvariant="bold">K</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>, and <inline-formula id="ieqn-162"><mml:math id="mml-ieqn-162"><mml:mrow><mml:mi mathvariant="bold">V</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>, where <inline-formula id="ieqn-163"><mml:math id="mml-ieqn-163"><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-164"><mml:math id="mml-ieqn-164"><mml:msub><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:msub></mml:math></inline-formula> denote the dimensionalities of the query/key and value vectors, respectively. The matrix <inline-formula id="ieqn-165"><mml:math id="mml-ieqn-165"><mml:mrow><mml:mi mathvariant="bold">Q</mml:mi></mml:mrow><mml:msup><mml:mrow><mml:mi mathvariant="bold">K</mml:mi></mml:mrow><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup></mml:math></inline-formula> computes pairwise similarity scores between queries and keys, and the softmax function is applied row-wise to normalize these scores across all key positions for each query, ensuring that attention weights sum to one. The scaling factor <inline-formula id="ieqn-166"><mml:math id="mml-ieqn-166"><mml:msqrt><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:msqrt></mml:math></inline-formula> prevents the dot products from becoming excessively large, improving numerical stability and gradient behavior. The resulting output is a weighted combination of value vectors, producing context-aware representations for each input position. <xref ref-type="fig" rid="fig-7">Fig. 7</xref> visualizes the scaled dot-product attention operation alongside its multi-head extension, showing how query, key, and value projections are combined to produce contextualized token representations.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Scaled dot-product self-attention mechanism [<xref ref-type="bibr" rid="ref-108">108</xref>]. Input token embeddings are linearly projected into query, key, and value representations. Attention weights are computed via normalized dot products between queries and keys, enabling each token to selectively aggregate information from all other tokens in the sequence. Adapted from [<xref ref-type="bibr" rid="ref-109">109</xref>].</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-7.tif"/>
</fig>
<p>Multi-head attention extends this by running <inline-formula id="ieqn-167"><mml:math id="mml-ieqn-167"><mml:mi>h</mml:mi></mml:math></inline-formula> parallel attention operations with different learned projections:<disp-formula id="eqn-22"><label>(22)</label><mml:math id="mml-eqn-22" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mtext>MultiHead</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">Q</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="bold">K</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="bold">V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mtext>Concat</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext>head</mml:mtext></mml:mrow><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext>head</mml:mtext></mml:mrow><mml:mi>h</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>O</mml:mi></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-168"><mml:math id="mml-ieqn-168"><mml:msub><mml:mrow><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">Q</mml:mi></mml:mrow><mml:msubsup><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>i</mml:mi><mml:mi>Q</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="bold">K</mml:mi></mml:mrow><mml:msubsup><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>i</mml:mi><mml:mi>K</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="bold">V</mml:mi></mml:mrow><mml:msubsup><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>i</mml:mi><mml:mi>V</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>. Here, <inline-formula id="ieqn-169"><mml:math id="mml-ieqn-169"><mml:mi>h</mml:mi></mml:math></inline-formula> denotes the number of attention heads, and <inline-formula id="ieqn-170"><mml:math id="mml-ieqn-170"><mml:msubsup><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>i</mml:mi><mml:mi>Q</mml:mi></mml:msubsup></mml:math></inline-formula>, <inline-formula id="ieqn-171"><mml:math id="mml-ieqn-171"><mml:msubsup><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>i</mml:mi><mml:mi>K</mml:mi></mml:msubsup></mml:math></inline-formula>, and <inline-formula id="ieqn-172"><mml:math id="mml-ieqn-172"><mml:msubsup><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>i</mml:mi><mml:mi>V</mml:mi></mml:msubsup></mml:math></inline-formula> are learnable projection matrices for the <inline-formula id="ieqn-173"><mml:math id="mml-ieqn-173"><mml:mi>i</mml:mi></mml:math></inline-formula>-th head. Each head operates on lower-dimensional subspaces, typically of size <inline-formula id="ieqn-174"><mml:math id="mml-ieqn-174"><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>h</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-175"><mml:math id="mml-ieqn-175"><mml:msub><mml:mi>d</mml:mi><mml:mi>v</mml:mi></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>h</mml:mi></mml:math></inline-formula>. The outputs of all heads are concatenated along the feature dimension using <inline-formula id="ieqn-176"><mml:math id="mml-ieqn-176"><mml:mrow><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">t</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> and then linearly transformed by a learnable projection matrix <inline-formula id="ieqn-177"><mml:math id="mml-ieqn-177"><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mi>O</mml:mi></mml:msub></mml:math></inline-formula> to produce the final output. This multi-head structure enables the model to capture diverse relationships by attending to information from different representation subspaces. Each head can attend to different aspects of the input, for sentiment analysis, some heads may focus on opinion words while others attend to negation or aspect terms. The overall encoder stack that forms the backbone of BERT and its variants is depicted in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>: each encoder block comprises multi-head self-attention followed by a position-wise feed-forward network, with residual connections and layer normalization applied at both sub-layers; BERT-base stacks 12 such blocks, while BERT-large extends this to 24, providing the representational capacity that underpins the performance gains reported on SST-2 and IMDb.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Transformer encoder architecture and BERT scaling [<xref ref-type="bibr" rid="ref-108">108</xref>]. Each encoder block consists of multi-head self-attention followed by position-wise feed-forward layers with residual connections and layer normalization. BERT-base stacks 12 such layers, while BERT-large extends this depth to 24 layers, enabling greater representational capacity for downstream sentiment analysis tasks. Adapted from [<xref ref-type="bibr" rid="ref-110">110</xref>].</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-8.tif"/>
</fig>
</sec>
<sec id="s4_4_2">
<label>4.4.2</label>
<title>BERT and Variants</title>
<p>BERT (Bidirectional Encoder Representations from Transformers) [<xref ref-type="bibr" rid="ref-16">16</xref>] pre-trains a deep transformer encoder using masked language modeling (MLM) and next sentence prediction (NSP) on large unlabeled corpora:<disp-formula id="eqn-23"><label>(23)</label><mml:math id="mml-eqn-23" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mtext>MLM</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:munder><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi>&#x02133;</mml:mi></mml:mrow></mml:mrow></mml:munder><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mrow><mml:mo>\</mml:mo><mml:mrow><mml:mi>&#x02133;</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-178"><mml:math id="mml-ieqn-178"><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi></mml:math></inline-formula> is the parameters of the transformer model (e.g., BERT) learned during pre-training. <inline-formula id="ieqn-179"><mml:math id="mml-ieqn-179"><mml:mrow><mml:mi>&#x02133;</mml:mi></mml:mrow></mml:math></inline-formula> denotes the set of masked token positions in the input sequence. The input <inline-formula id="ieqn-180"><mml:math id="mml-ieqn-180"><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow></mml:math></inline-formula> is first corrupted by replacing tokens at positions in <inline-formula id="ieqn-181"><mml:math id="mml-ieqn-181"><mml:mrow><mml:mi>&#x02133;</mml:mi></mml:mrow></mml:math></inline-formula> with a special [MASK] token (or, with small probability, random or unchanged tokens), and the model is trained to predict the original tokens <inline-formula id="ieqn-182"><mml:math id="mml-ieqn-182"><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> at those positions. Accordingly, <inline-formula id="ieqn-183"><mml:math id="mml-ieqn-183"><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mrow><mml:mo>\</mml:mo><mml:mrow><mml:mi>&#x02133;</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> denotes the conditional probability of the original token given the corrupted input sequence, where masked positions remain present but altered rather than removed. This objective can be interpreted as maximizing the log-likelihood of masked tokens under a corruption distribution over input sequences.</p>
<p>BERT-base (110M parameters) achieved 94.9% on SST-2 upon release, representing a dramatic improvement over prior methods [<xref ref-type="bibr" rid="ref-16">16</xref>]. Several subsequent variants optimized different aspects of the pre-training and architecture. RoBERTa [<xref ref-type="bibr" rid="ref-17">17</xref>] removed NSP, trained with larger batches and more data, and used dynamic masking, pushing SST-2 accuracy to 96.4%. XLNet [<xref ref-type="bibr" rid="ref-18">18</xref>] combined autoregressive modeling with permutation-based training to capture bidirectional context without masking artifacts. DistilBERT [<xref ref-type="bibr" rid="ref-111">111</xref>] applied knowledge distillation to compress BERT to 66M parameters (40% smaller) while retaining 97% of its performance, enabling deployment in resource-constrained settings.</p>
<p>For domain-specific applications, Sun et al. [<xref ref-type="bibr" rid="ref-19">19</xref>] established best practices for fine-tuning BERT on sentiment tasks, finding that gradual unfreezing and discriminative learning rates significantly improve convergence. Bello et al. [<xref ref-type="bibr" rid="ref-112">112</xref>] and Batra et al. [<xref ref-type="bibr" rid="ref-113">113</xref>] demonstrated BERT&#x2019;s effectiveness for tweet and software review sentiment analysis, respectively, while Penha and Hauff [<xref ref-type="bibr" rid="ref-114">114</xref>] probed BERT&#x2019;s internal representations to understand what it learns about domain-specific sentiment. InstructABSA [<xref ref-type="bibr" rid="ref-80">80</xref>] takes a different tack, reformulating ABSA as an instruction-following task for instruction-tuned transformers and achieving competitive results with minimal task-specific fine-tuning.</p>
</sec>
<sec id="s4_4_3">
<label>4.4.3</label>
<title>ModernBERT</title>
<p>Warner et al. [<xref ref-type="bibr" rid="ref-58">58</xref>] introduced ModernBERT in December 2024 as a modernized encoder architecture that incorporates long-context training, rotary positional embeddings (RoPE), and optimized attention kernels. The key innovations include an extended context length of 8192 tokens (compared to BERT&#x2019;s 512), enabling processing of full-length reviews without truncation; training on 2 trillion tokens, an order of magnitude more than the original BERT; rotary positional embeddings (RoPE) replacing absolute position encodings to improve length generalization; and Flash Attention with unpadding for efficient inference. ModernBERT outperforms BERT, RoBERTa, and DeBERTa across classification tasks while maintaining efficient inference, making it a strong candidate for production SA systems that require the fine-tuned model paradigm. Recent work combining ModernBERT with SHAP and LIME has further demonstrated enhanced explainability for sentiment classification [<xref ref-type="bibr" rid="ref-68">68</xref>].</p>
</sec>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Large Language Models</title>
<p>The emergence of large language models has introduced a qualitatively new paradigm in which models can perform sentiment analysis without task-specific fine-tuning, relying instead on zero-shot prompting, few-shot in-context learning, and chain-of-thought reasoning.</p>
<sec id="s4_5_1">
<label>4.5.1</label>
<title>Zero-Shot and Few-Shot Sentiment Analysis</title>
<p>LLMs leverage their pre-training on massive corpora to perform SA through natural language instructions. In zero-shot mode, the model receives only a task description such as &#x201C;<italic>Classify the sentiment of the following review as positive or negative: [review text]</italic>.&#x201D; In few-shot mode, <inline-formula id="ieqn-184"><mml:math id="mml-ieqn-184"><mml:mi>k</mml:mi></mml:math></inline-formula> labeled examples are prepended as context [<xref ref-type="bibr" rid="ref-20">20</xref>,<xref ref-type="bibr" rid="ref-115">115</xref>]:<disp-formula id="eqn-24"><label>(24)</label><mml:math id="mml-eqn-24" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D49F;</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>;</mml:mo><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>P</mml:mi><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow></mml:mstyle><mml:mi>y</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">]</mml:mo><mml:mo>;</mml:mo><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mstyle scriptlevel="0"><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-185"><mml:math id="mml-ieqn-185"><mml:msub><mml:mrow><mml:mi>&#x1D49F;</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:msubsup><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>k</mml:mi></mml:msubsup></mml:math></inline-formula> denotes the set of demonstration examples, and <inline-formula id="ieqn-186"><mml:math id="mml-ieqn-186"><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> represents the concatenation of input-output pairs into a single prompt sequence. The probability is computed by a pre-trained language model with fixed parameters <inline-formula id="ieqn-187"><mml:math id="mml-ieqn-187"><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi></mml:math></inline-formula> via autoregressive token prediction.</p>
<p>Zhang et al. [<xref ref-type="bibr" rid="ref-22">22</xref>] conducted the most comprehensive evaluation to date using SentiEval, testing across 13 tasks and 26 datasets. Their findings reveal a nuanced picture: in zero-shot settings, LLMs such as ChatGPT and GPT-4 achieve satisfactory performance on simple binary classification but lag behind fine-tuned SLMs on complex tasks such as ABSA and fine-grained classification; in few-shot settings, LLMs significantly outperform SLMs, suggesting particular value when annotation resources are limited; and prompt design has a substantial impact, with performance variance across five different prompts reaching up to 15 percentage points on some datasets. To summarize these trends visually, <xref ref-type="fig" rid="fig-9">Fig. 9</xref> presents an illustrative, survey-level comparison of zero-shot and fine-tuned sentiment analysis performance across representative model paradigms.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>Illustrative comparison of zero-shot (prompted) and task-specific fine-tuned relative performance (%) across major sentiment analysis model paradigms. Bar heights reflect normalized trends consistently reported across multiple benchmarks and studies, rather than absolute accuracy values from a single dataset or experimental configuration (This figure was generated with the assistance of OpenAI GPT 5.2).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-9.tif"/>
</fig>
<p>Rathje et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] evaluated GPT-3.5, GPT-4, and GPT-4-Turbo on 15 psychological text analysis datasets across 12 languages, finding correlations in the range <inline-formula id="ieqn-188"><mml:math id="mml-ieqn-188"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.59</mml:mn></mml:math></inline-formula>&#x2013;0.77 with human annotations, compared to <inline-formula id="ieqn-189"><mml:math id="mml-ieqn-189"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.20</mml:mn></mml:math></inline-formula>&#x2013;0.30 for dictionary-based methods, a dramatic improvement for multilingual SA. Krugmann and Hartmann [<xref ref-type="bibr" rid="ref-23">23</xref>] provided a systematic analysis of generative AI&#x2019;s potential for SA, identifying prompt engineering, cost efficiency, and hallucination risk as key considerations for practical deployment.</p>
</sec>
<sec id="s4_5_2">
<label>4.5.2</label>
<title>Open-Source vs. Closed-Source LLMs</title>
<p>The LLM landscape for sentiment analysis is bifurcated between proprietary models and open-source alternatives. On the open-source side, Meta&#x2019;s Llama-3 [<xref ref-type="bibr" rid="ref-21">21</xref>] family culminated in the Llama-3.1-405B model, which Meta reports was pretrained on approximately 15 trillion tokens and employs grouped query attention (GQA) for efficient inference; it demonstrates strong performance on SA benchmarks, particularly when fine-tuned with domain-specific data. On the proprietary side, GPT-4 and GPT-4o represent the current state of the art; Shen and Zhang [<xref ref-type="bibr" rid="ref-52">52</xref>] showed that GPT-4o with few-shot prompting matches FinBERT&#x2019;s fine-tuned performance on financial news sentiment, while Bhatia et al. [<xref ref-type="bibr" rid="ref-87">87</xref>] demonstrated that FinTral, a multimodal financial LLM family built on Llama 3, achieves GPT-4-level performance on financial SA tasks.</p>
<p>A comprehensive comparison by Zhu et al. [<xref ref-type="bibr" rid="ref-62">62</xref>] in their Model Arena for Cross-lingual Sentiment Analysis evaluated XLM-R, Llama-3, and GPT-4, finding that GPT-4 maintains a consistent advantage in zero-shot cross-lingual settings, but the gap narrows substantially when open-source models are fine-tuned on target-language data. The most recent human-vs.-LLM comparison study [<xref ref-type="bibr" rid="ref-53">53</xref>] tested 33 human annotators against 8 LLM variants (GPT-3.5, GPT-4, GPT-4o, Gemini, Llama-3.1, Mixtral) on 100 items, finding that GPT-4o achieved the highest agreement with expert consensus.</p>
</sec>
<sec id="s4_5_3">
<label>4.5.3</label>
<title>Chain-of-Thought and Reasoning-Augmented Sentiment Analysis</title>
<p>A particularly promising development is the use of chain-of-thought (CoT) prompting [<xref ref-type="bibr" rid="ref-116">116</xref>] to improve LLM performance on sentiment tasks that require implicit reasoning. Fei et al. [<xref ref-type="bibr" rid="ref-54">54</xref>] introduced <bold>THOR</bold> (Three-Hop Reasoning), a structured CoT framework specifically designed for implicit sentiment analysis. The three hops consist of identifying the stimulating aspect, inferring the opinion holder&#x2019;s emotional reaction, and deriving the sentiment polarity. Formally, the reasoning chain augments the prediction:<disp-formula id="eqn-25"><label>(25)</label><mml:math id="mml-eqn-25" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo>;</mml:mo><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:munder><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mrow><mml:mi>&#x211B;</mml:mi></mml:mrow></mml:mrow></mml:munder><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi>&#x211B;</mml:mi></mml:mrow><mml:mo>;</mml:mo><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x211B;</mml:mi></mml:mrow><mml:mo>&#x2223;</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo>;</mml:mo><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-190"><mml:math id="mml-ieqn-190"><mml:mrow><mml:mi>&#x211B;</mml:mi></mml:mrow></mml:math></inline-formula> denotes a reasoning chain (e.g., a sequence of intermediate reasoning steps generated via chain-of-thought prompting). This formulation interprets reasoning as a latent variable that mediates the prediction. In practice, however, large language models approximate this marginalization by generating one or a small number of reasoning paths rather than explicitly summing over all possible <inline-formula id="ieqn-191"><mml:math id="mml-ieqn-191"><mml:mrow><mml:mi>&#x211B;</mml:mi></mml:mrow></mml:math></inline-formula>.</p>
<p>THOR demonstrated significant improvements on implicit sentiment datasets where direct classification fails. Duan and Wang [<xref ref-type="bibr" rid="ref-55">55</xref>] proposed a complementary framework called <bold>SAoT</bold> (Sentiment Analysis of Thought) using ERNIE-Bot-4, which decomposes implicit sentiment analysis into structured reasoning steps with explicit intermediate outputs, achieving improvements over direct prompting approaches.</p>
<p>However, Zheng et al. [<xref ref-type="bibr" rid="ref-56">56</xref>] provide an important critical reassessment, demonstrating that CoT does not uniformly improve SA performance across all task types. Their analysis suggests that CoT is most beneficial for tasks requiring pragmatic inference (implicit sentiment, sarcasm) but can actually hurt performance on straightforward polarity classification, where the reasoning overhead introduces unnecessary error propagation. Recent extensions include multi-chain CoT for ABSA [<xref ref-type="bibr" rid="ref-57">57</xref>], which employs multiple parallel reasoning chains to handle the complexity of aspect-level sentiment, and graph-enhanced reasoning [<xref ref-type="bibr" rid="ref-117">117</xref>], which combines graph neural networks with prompt-based reasoning for implicit aspect-based sentiment analysis.</p>
</sec>
<sec id="s4_5_4">
<label>4.5.4</label>
<title>Model Variability Problem</title>
<p>Herrera-Poyatos et al. [<xref ref-type="bibr" rid="ref-25">25</xref>] identified the <italic>model variability problem</italic> (MVP) as a fundamental challenge for LLM-based sentiment analysis. MVP manifests as inconsistent sentiment classifications for identical inputs across multiple inference runs, arising from three principal sources. First, stochastic decoding with temperature-based sampling (<inline-formula id="ieqn-192"><mml:math id="mml-ieqn-192"><mml:mi>T</mml:mi><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula>) introduces randomness into token selection, where the probability of generating token <inline-formula id="ieqn-193"><mml:math id="mml-ieqn-193"><mml:mi>w</mml:mi></mml:math></inline-formula> is modulated as:<disp-formula id="eqn-26"><label>(26)</label><mml:math id="mml-eqn-26" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>w</mml:mi><mml:mo>&#x2223;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi mathvariant="bold-italic">&#x03B8;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mi>w</mml:mi></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>T</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:msup><mml:mi>w</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:mrow></mml:msub><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mrow><mml:msup><mml:mi>w</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:mrow></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>T</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-194"><mml:math id="mml-ieqn-194"><mml:msub><mml:mi>z</mml:mi><mml:mi>w</mml:mi></mml:msub></mml:math></inline-formula> denotes the logit corresponding to token <inline-formula id="ieqn-195"><mml:math id="mml-ieqn-195"><mml:mi>w</mml:mi></mml:math></inline-formula> produced by the model given the previous context <inline-formula id="ieqn-196"><mml:math id="mml-ieqn-196"><mml:msub><mml:mrow><mml:mi mathvariant="bold">w</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, and <inline-formula id="ieqn-197"><mml:math id="mml-ieqn-197"><mml:mi>T</mml:mi><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> is the temperature parameter controlling the sharpness of the distribution. Lower values of <italic>T</italic> concentrate probability mass on high-scoring tokens, while higher values increase output diversity. Second, prompt sensitivity means that minor rephrasing of prompts can shift predictions across decision boundaries. Third, systematic biases in pre-training corpora propagate into inconsistent sentiment judgments. The authors emphasize the role of explainability frameworks in mitigating MVP, connecting to broader XAI efforts discussed in <xref ref-type="sec" rid="s6_5">Section 6.5</xref>.</p>
<p>To quantify MVP in a standardized manner, several measurement approaches can be defined. The <italic>classification instability rate</italic> (CIR) measures the proportion of inputs that receive different sentiment labels across <inline-formula id="ieqn-198"><mml:math id="mml-ieqn-198"><mml:mi>k</mml:mi></mml:math></inline-formula> repeated inference runs:<disp-formula id="eqn-27"><label>(27)</label><mml:math id="mml-eqn-27" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mtext>CIR</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mi mathvariant="normal">&#x2203;</mml:mi><mml:mspace width="thinmathspace" /><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo>:</mml:mo><mml:msubsup><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup><mml:mo>&#x2260;</mml:mo><mml:msubsup><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-199"><mml:math id="mml-ieqn-199"><mml:msubsup><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> denotes the predicted label for input <inline-formula id="ieqn-200"><mml:math id="mml-ieqn-200"><mml:mi>i</mml:mi></mml:math></inline-formula> on run <inline-formula id="ieqn-201"><mml:math id="mml-ieqn-201"><mml:mi>j</mml:mi></mml:math></inline-formula>. <inline-formula id="ieqn-202"><mml:math id="mml-ieqn-202"><mml:mrow><mml:mn mathvariant="bold">1</mml:mn></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> denotes the indicator function, which equals 1 if its condition is true and 0 otherwise, and <inline-formula id="ieqn-203"><mml:math id="mml-ieqn-203"><mml:mi>k</mml:mi></mml:math></inline-formula> is the number of repeated inference runs for the same input. Thus, <inline-formula id="ieqn-204"><mml:math id="mml-ieqn-204"><mml:mrow><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">I</mml:mi><mml:mi mathvariant="normal">R</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula>, with <inline-formula id="ieqn-205"><mml:math id="mml-ieqn-205"><mml:mrow><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">I</mml:mi><mml:mi mathvariant="normal">R</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> indicating perfectly stable classifications across runs and <inline-formula id="ieqn-206"><mml:math id="mml-ieqn-206"><mml:mrow><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">I</mml:mi><mml:mi mathvariant="normal">R</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> indicating that every input receives at least one inconsistent label across the <inline-formula id="ieqn-207"><mml:math id="mml-ieqn-207"><mml:mi>k</mml:mi></mml:math></inline-formula> runs. Complementarily, entropy-based variability captures the spread of output distributions: for each input <inline-formula id="ieqn-208"><mml:math id="mml-ieqn-208"><mml:msub><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula>, the empirical entropy <inline-formula id="ieqn-209"><mml:math id="mml-ieqn-209"><mml:msub><mml:mi>H</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi>&#x1D49E;</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mover><mml:mi>p</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>c</mml:mi></mml:msub><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>p</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula> over classes, averaged across the dataset, quantifies overall model uncertainty. The SentiEval evaluation [<xref ref-type="bibr" rid="ref-22">22</xref>] provides indirect evidence of prompt-induced variability, documenting 10%&#x2013;15% performance swings across five different prompt formulations on the same dataset.</p>
<p>It is important to distinguish between <italic>aleatoric</italic> variability (inherent in stochastic decoding at <inline-formula id="ieqn-210"><mml:math id="mml-ieqn-210"><mml:mi>T</mml:mi><mml:mo>&#x003E;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula>, reducible by setting <inline-formula id="ieqn-211"><mml:math id="mml-ieqn-211"><mml:mi>T</mml:mi><mml:mo stretchy="false">&#x2192;</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> or using greedy decoding) and <italic>epistemic</italic> variability (arising from prompt sensitivity and training data biases, which persists even at <inline-formula id="ieqn-212"><mml:math id="mml-ieqn-212"><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula>). Current mitigation strategies ensemble averaging over multiple runs, temperature annealing, majority voting are empirically motivated but lack formal convergence guarantees. Developing principled methods to bound and reduce MVP remains an important open problem with direct implications for deployment in regulated industries such as finance and healthcare.</p>
</sec>
<sec id="s4_5_5">
<label>4.5.5</label>
<title>Prompt Engineering for Sentiment Analysis</title>
<p>Effective prompt design is critical for LLM-based SA. Stilwell and Inkpen [<xref ref-type="bibr" rid="ref-118">118</xref>] show that prompt wording significantly affects model performance on the IMDb sentiment benchmark and that prompt learning can substantially improve classification accuracy. They also investigate explainability methods for prompt-based classifiers, finding that explanations generated directly by LLMs are judged by humans to be more adequate and trustworthy than traditional XAI approaches such as LIME or SHAP. The prompts evaluated in their study include instructional, completion, and question-based formulations that guide the model to produce binary sentiment labels. Gu et al. [<xref ref-type="bibr" rid="ref-119">119</xref>] introduced prompt tuning as an alternative to manual prompt engineering, learning continuous prompt vectors that are prepended to the input and achieving few-shot performance comparable to full fine-tuning. More systematically, prompting strategies for SA can be organized into a hierarchy of increasing sophistication: (1) <italic>zero-shot instruction prompts</italic>, which provide only the task description and output format which are effective for simple binary classification but insufficient for nuanced tasks; (2) <italic>few-shot demonstration prompts</italic>, which prepend labeled examples, particularly effective when demonstrations are drawn from the target domain and cover diverse sentiment patterns; (3) <italic>chain-of-thought prompts</italic>, which request step-by-step reasoning before classification, most beneficial for implicit sentiment and sarcasm, but potentially harmful for straightforward polarity tasks due to error propagation [<xref ref-type="bibr" rid="ref-56">56</xref>]; and (4) <italic>role-specification prompts</italic>. <xref ref-type="fig" rid="fig-10">Fig. 10</xref> illustrates how variations in prompt formulation can lead to different sentiment interpretations for the same input text, even when using the same underlying language model.</p>
<fig id="fig-10">
<label>Figure 10</label>
<caption>
<title>Illustration of prompt sensitivity in large language model&#x2013;based sentiment analysis. The same input text can yield different sentiment interpretations depending on prompt formulation, highlighting the dependence of LLM-based sentiment outputs on task framing and instruction design (This figure was generated with the assistance of OpenAI GPT 5.2).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-10.tif"/>
</fig>
</sec>
<sec id="s4_5_6">
<label>4.5.6</label>
<title>Ensemble Approaches</title>
<p>A comprehensive survey of ensemble strategies for LLMs [<xref ref-type="bibr" rid="ref-71">71</xref>] catalogs methods for combining multiple LLM outputs to improve SA robustness, including majority voting across model variants, weighted aggregation based on confidence scores, and mixture-of-experts architectures that route inputs to specialized sub-models. Ensemble approaches directly address the MVP by averaging over stochastic outputs, though at increased computational cost.</p>
</sec>
</sec>
<sec id="s4_6">
<label>4.6</label>
<title>Multimodal Sentiment Analysis</title>
<p>Human communication is inherently multimodal, conveying sentiment through not only words but also tone of voice, facial expressions, and gestures [<xref ref-type="bibr" rid="ref-34">34</xref>]. Multimodal sentiment analysis (MSA) integrates these complementary signals, and the choice of fusion strategy, that is, how modality-specific representations are combined, is a central design decision.</p>
<sec id="s4_6_1">
<label>4.6.1</label>
<title>Fusion Strategies</title>
<p>Given modality-specific representations <inline-formula id="ieqn-213"><mml:math id="mml-ieqn-213"><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> (text), <inline-formula id="ieqn-214"><mml:math id="mml-ieqn-214"><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>a</mml:mi></mml:msub></mml:math></inline-formula> (audio), and <inline-formula id="ieqn-215"><mml:math id="mml-ieqn-215"><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>v</mml:mi></mml:msub></mml:math></inline-formula> (visual), fusion strategies determine how these are combined [<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>,<xref ref-type="bibr" rid="ref-91">91</xref>].</p>
<p><bold>Early fusion</bold> concatenates raw or lightly processed features before a joint model:<disp-formula id="eqn-28"><label>(28)</label><mml:math id="mml-eqn-28" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mtext>fused</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>a</mml:mi></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>v</mml:mi></mml:msub><mml:mo stretchy="false">]</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where, <inline-formula id="ieqn-216"><mml:math id="mml-ieqn-216"><mml:mo stretchy="false">[</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo>;</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo>;</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">]</mml:mo></mml:math></inline-formula> denotes vector concatenation along the feature dimension.</p>
<p><bold>Late fusion</bold> trains separate modality-specific classifiers and combines their predictions:<disp-formula id="eqn-29"><label>(29)</label><mml:math id="mml-eqn-29" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>a</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>a</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>v</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>v</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-217"><mml:math id="mml-ieqn-217"><mml:mi>g</mml:mi></mml:math></inline-formula> is a combination function (e.g., weighted average, learned MLP). <inline-formula id="ieqn-218"><mml:math id="mml-ieqn-218"><mml:msub><mml:mi>f</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-219"><mml:math id="mml-ieqn-219"><mml:msub><mml:mi>f</mml:mi><mml:mi>a</mml:mi></mml:msub></mml:math></inline-formula>, and <inline-formula id="ieqn-220"><mml:math id="mml-ieqn-220"><mml:msub><mml:mi>f</mml:mi><mml:mi>v</mml:mi></mml:msub></mml:math></inline-formula> are modality-specific prediction functions (e.g., classifiers), and <inline-formula id="ieqn-221"><mml:math id="mml-ieqn-221"><mml:mi>g</mml:mi></mml:math></inline-formula> is a fusion function that combines their outputs.</p>
<p><bold>Tensor fusion</bold> [<xref ref-type="bibr" rid="ref-35">35</xref>] computes the outer product across modalities to capture multiplicative interactions:<disp-formula id="eqn-30"><label>(30)</label><mml:math id="mml-eqn-30" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mi mathvariant="bold">z</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x2297;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>a</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x2297;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mi>v</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>The appended 1 captures unimodal and bimodal interactions as subsets of the full tensor. Here, each modality vector is augmented with a constant 1, and <inline-formula id="ieqn-222"><mml:math id="mml-ieqn-222"><mml:mo>&#x2297;</mml:mo></mml:math></inline-formula> denotes the outer (tensor) product, enabling the model to capture unimodal, bimodal, and trimodal interactions. While tensor fusion is theoretically expressive, it suffers from exponential dimensionality growth.</p>
<p><bold>Cross-modal attention</bold> [<xref ref-type="bibr" rid="ref-90">90</xref>] learns which elements of one modality are most relevant to another:<disp-formula id="eqn-31"><label>(31)</label><mml:math id="mml-eqn-31" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="bold">h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo stretchy="false">&#x2190;</mml:mo><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mtext>Attention</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">Q</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">K</mml:mi></mml:mrow><mml:mi>a</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">V</mml:mi></mml:mrow><mml:mi>a</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>Here, <inline-formula id="ieqn-223"><mml:math id="mml-ieqn-223"><mml:msub><mml:mrow><mml:mi mathvariant="bold">Q</mml:mi></mml:mrow><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-224"><mml:math id="mml-ieqn-224"><mml:msub><mml:mrow><mml:mi mathvariant="bold">K</mml:mi></mml:mrow><mml:mi>a</mml:mi></mml:msub></mml:math></inline-formula>, and <inline-formula id="ieqn-225"><mml:math id="mml-ieqn-225"><mml:msub><mml:mrow><mml:mi mathvariant="bold">V</mml:mi></mml:mrow><mml:mi>a</mml:mi></mml:msub></mml:math></inline-formula> denote query, key, and value projections derived from the text and audio modalities, respectively.</p>
<p>Multi-attention recurrent networks [<xref ref-type="bibr" rid="ref-90">90</xref>] apply this bidirectionally across all modality pairs. More recently, Cai et al. [<xref ref-type="bibr" rid="ref-120">120</xref>] proposed a multi-layer feature fusion approach combined with multi-task learning, demonstrating improvements on CMU-MOSI and CMU-MOSEI, while Ren [<xref ref-type="bibr" rid="ref-121">121</xref>] combined BERT for text encoding with ResNet for image features, achieving 74.5% accuracy on the MAVA-single multimodal benchmark through an attention-based fusion mechanism. <xref ref-type="fig" rid="fig-11">Fig. 11</xref> summarizes the principal multimodal fusion paradigms used in sentiment analysis, highlighting the trade-offs between representational expressiveness and computational cost.</p>
<fig id="fig-11">
<label>Figure 11</label>
<caption>
<title>Taxonomy of multimodal fusion strategies for sentiment analysis. (<bold>a</bold>) Early fusion concatenates modality representations before joint modeling. (<bold>b</bold>) Late fusion trains independent classifiers per modality and combines predictions. (<bold>c</bold>) Tensor fusion computes outer products across modalities at exponential dimensionality cost. (<bold>d</bold>) Cross-modal attention enables selective information flow between modality pairs through query&#x2013;key&#x2013;value projections (This figure was generated with the assistance of OpenAI GPT 5.2).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-11.tif"/>
</fig>
</sec>
<sec id="s4_6_2">
<label>4.6.2</label>
<title>Multimodal Transformers</title>
<p>Transformer architectures have been adapted for multimodal inputs, with MMBERT [<xref ref-type="bibr" rid="ref-122">122</xref>] extending BERT&#x2019;s architecture with cross-modal attention layers that fuse text and visual representations. More recently, the explosive growth of multimodal LLMs (MLLMs) has opened new possibilities for sentiment analysis. Yang et al. [<xref ref-type="bibr" rid="ref-36">36</xref>] provide a comprehensive survey of how LLMs are being integrated into text-centric multimodal SA, identifying key challenges in grounding visual and acoustic features within LLM representations. da Silva et al. [<xref ref-type="bibr" rid="ref-59">59</xref>] introduce MLLMsent, a framework for evaluating whether multimodal LLMs can &#x201C;see&#x201D; sentiment in images and text, achieving state-of-the-art results that significantly outperform prior specialist models; their findings suggest that MLLMs&#x2019; ability to jointly reason over visual and textual modalities provides a natural advantage for tasks that require understanding both content and presentation. A systematic review [<xref ref-type="bibr" rid="ref-60">60</xref>], analyzing 116 multimodal SA studies published between 2018 and 2025, identifies three key trends: the shift from handcrafted fusion to end-to-end learned fusion, the growing dominance of transformer-based architectures, and the persistent challenge of temporal alignment between modalities.</p>
<p>A critical architectural distinction in recent multimodal SA is between <italic>early-fusion multimodal transformers</italic>, which tokenize and embed all modalities into a shared sequence space from the initial layers (enabling deep cross-modal interaction but requiring modality-specific tokenizers), and <italic>late-integration approaches</italic>, which employ modality-specific encoders followed by cross-modal attention in upper layers (preserving modality-specific representations but potentially limiting deep interaction). The MLLMsent results [<xref ref-type="bibr" rid="ref-59">59</xref>] suggest that general-purpose multimodal LLMs, which typically use the late-integration paradigm, can leverage their massive pre-training to overcome the interaction depth limitation, achieving competitive results without task-specific architectural design. However, the systematic review [<xref ref-type="bibr" rid="ref-60">60</xref>] notes that text features continue to dominate in most fusion architectures, raising the question of whether current models truly leverage multimodal information or primarily rely on text with minor acoustic/visual augmentation.</p>
</sec>
</sec>
<sec id="s4_7">
<label>4.7</label>
<title>Comparative Analysis across Paradigms</title>
<p><xref ref-type="table" rid="table-3">Table 3</xref> provides a detailed comparison across representative models from each paradigm on standard benchmarks. Several patterns emerge from this comparison. First, there is a consistent upward trajectory in performance, with each paradigm shift yielding measurable improvements. Second, the gap between fine-tuned transformers and zero-shot LLMs is surprisingly small on binary classification tasks but widens substantially for fine-grained and aspect-based tasks [<xref ref-type="bibr" rid="ref-22">22</xref>]. Third, the computational cost increases by orders of magnitude across paradigms: from negligible for lexicon methods to billions of FLOPs for LLM inference, creating a trade-off where practitioners must balance accuracy against computational budget [<xref ref-type="bibr" rid="ref-123">123</xref>].</p>
<p>The paradox of the current era is that LLMs are simultaneously the most capable and the most unreliable SA systems: they achieve near-human performance on many tasks [<xref ref-type="bibr" rid="ref-53">53</xref>] while suffering from the model variability problem [<xref ref-type="bibr" rid="ref-25">25</xref>], prompt sensitivity [<xref ref-type="bibr" rid="ref-22">22</xref>], and occasional hallucination [<xref ref-type="bibr" rid="ref-23">23</xref>]. This makes the choice between fine-tuned SLMs and prompted LLMs a non-trivial engineering decision that depends heavily on the specific deployment context. <xref ref-type="fig" rid="fig-12">Fig. 12</xref> visualizes this performance&#x2013;cost trade-off across representative sentiment analysis model paradigms. it is worth mentioning that <xref ref-type="fig" rid="fig-12">Fig. 12</xref> is a survey-level synthesis based on relative trends reported across multiple benchmarks and studies, rather than absolute results from a single dataset or experimental setup.</p>
<fig id="fig-12">
<label>Figure 12</label>
<caption>
<title>Illustrative performance&#x2013;cost trade-offs across major sentiment analysis model paradigms. Relative sentiment analysis performance (%) is shown against normalized computational cost (arbitrary units) for training and inference, summarizing trends reported across multiple benchmarks and studies rather than absolute results from a single experimental setup (This figure was generated with the assistance of OpenAI GPT 5.2).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-12.tif"/>
</fig>
<p>While successive methodological paradigms have delivered substantial performance gains on standard benchmarks, these advances have also revealed new limitations that become evident in real-world deployment. High accuracy on curated datasets does not guarantee robustness to sarcasm, domain drift, cultural variation, or adversarial content. The next section examines these domain-specific challenges in detail, highlighting persistent failure modes that remain unresolved even in the era of large language models.</p>
<sec id="s4_7_1">
<label>4.7.1</label>
<title>Failure Modes and Applicability Boundaries</title>
<p>A deeper analysis reveals paradigm-specific failure modes that determine applicability boundaries. Lexicon-based methods fail systematically on compositional semantics (&#x201C;not bad&#x201D; <inline-formula id="ieqn-226"><mml:math id="mml-ieqn-226"><mml:mo stretchy="false">&#x2192;</mml:mo></mml:math></inline-formula> misclassified as negative), figurative language, and any domain where the lexicon lacks coverage: they are best suited for quick, interpretable baselines in well-understood domains. Classical ML methods (SVM, Na&#x00EF;ve Bayes) are brittle to out-of-vocabulary terms and domain shift, as their bag-of-words representations cannot generalize beyond the training vocabulary; they remain appropriate for resource-constrained settings with stable, in-domain data. Deep learning models (CNN, LSTM) degrade on short texts where sequential patterns are sparse and on aspect-level tasks without sufficient aspect-annotated training data; they offer the best accuracy-to-cost ratio for medium-scale supervised settings. Fine-tuned transformers (BERT, ModernBERT) achieve near-ceiling performance on standard benchmarks but struggle with implicit sentiment and sarcasm (where surface-level patterns are misleading) and require per-task fine-tuning that scales poorly across domains; they are the current best choice for production systems with well-defined task boundaries. LLMs excel at zero-shot generalization and few-shot adaptation but suffer from the MVP, prompt sensitivity, hallucinated sentiment labels (particularly for ambiguous inputs), and high inference cost; they are most valuable for low-resource, multi-domain, or rapidly evolving deployment contexts where fine-tuning is impractical.</p>
</sec>
<sec id="s4_7_2">
<label>4.7.2</label>
<title>Practitioner Decision Framework</title>
<p>The choice among SA paradigms involves navigating a multi-dimensional trade-off space. Based on the empirical patterns documented throughout this survey, we propose the following structured decision framework for practitioners:</p>
<p>Task complexity: For binary classification on well-defined domains, fine-tuned transformers (BERT/ModernBERT) offer the best accuracy-to-cost ratio. For ABSA and fine-grained tasks, fine-tuned SLMs outperform zero-shot LLMs [<xref ref-type="bibr" rid="ref-22">22</xref>]. For implicit sentiment and tasks requiring pragmatic inference, CoT-prompted LLMs show advantages [<xref ref-type="bibr" rid="ref-54">54</xref>,<xref ref-type="bibr" rid="ref-55">55</xref>].</p>
<p>Available labeled data: With &#x003E;1K labeled examples per class, fine-tuning is generally superior. With 5&#x2013;50 examples, few-shot LLMs become competitive. With zero labeled examples, LLM zero-shot is the only viable option among neural methods, though VADER remains a useful baseline for social media text.</p>
<p>Latency and throughput: Lexicon methods process thousands of documents per second. Inference throughput for transformer classifiers such as BERT varies substantially depending on hardware configuration, sequence length, and batching strategy. Reported performance in practical deployments ranges from tens to thousands of documents per second on modern GPUs. Inference latency for LLM-based sentiment analysis depends strongly on model size, provider infrastructure, and prompt length. Reported response times for API-based systems typically range from sub-second to several seconds per request.</p>
<p>Consistency requirements: For regulated industries (finance, healthcare) where reproducibility is mandated, the MVP makes raw LLM deployment problematic. Ensemble methods [<xref ref-type="bibr" rid="ref-71">71</xref>] or deterministic decoding (<inline-formula id="ieqn-227"><mml:math id="mml-ieqn-227"><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula>) can partially mitigate this, but fine-tuned SLMs with fixed weights offer inherently deterministic inference.</p>
<p>Computational budget: Fine-tuning compact transformer models such as BERT-base can often be completed within a few GPU-hours for moderate-sized datasets, although training time varies significantly with dataset scale, model configuration, and hardware. Token-based pricing for commercial LLM APIs varies widely across providers and model classes, typically ranging from fractions of a cent to several cents per thousand tokens depending on model capability and deployment tier. The Green AI framework [<xref ref-type="bibr" rid="ref-123">123</xref>] advocates reporting computational costs alongside accuracy to enable informed paradigm selection.</p>
<p>A consistent pattern emerges across the methodological evolution of sentiment analysis: performance improvements on standard benchmarks such as SST-2 and IMDb have largely saturated with transformer-based models, while more complex tasks continue to expose fundamental limitations. In particular, fine-grained sentiment classification, aspect-based sentiment analysis, and cross-lingual transfer reveal a widening gap between model architectures. While large language models demonstrate strong zero-shot capabilities, they remain less reliable than fine-tuned transformer models in structured tasks, especially under domain shift. This suggests that recent progress is no longer driven primarily by raw model capacity, but by the ability to balance generalization, stability, and task-specific adaptation.</p>
</sec>
<sec id="s4_7_3">
<label>4.7.3</label>
<title>Synthesis of Paradigm Trade-Offs</title>
<p>While prior comparisons (<xref ref-type="table" rid="table-2">Tables 2</xref> and <xref ref-type="table" rid="table-3">3</xref>) emphasize benchmark performance, a broader pattern emerges when sentiment analysis paradigms are evaluated across qualitative capability dimensions. Specifically, recent progress is characterized less by absolute accuracy gains and more by trade-offs between contextual understanding, generalization ability, interpretability, and stability.</p>
<p><xref ref-type="table" rid="table-4">Table 4</xref> reveals that the evolution of sentiment analysis has shifted from improving accuracy to navigating trade-offs between generalization, interpretability, and robustness. In particular, while LLMs offer strong zero-shot capabilities, their instability and lack of interpretability present significant challenges for deployment, whereas transformer-based models remain more reliable in controlled settings.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Analytical comparison of sentiment analysis paradigms across capability dimensions. Unlike performance-focused comparisons, this table highlights structural trade-offs between paradigms.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th>Paradigm</th>
<th>Context Sensitivity</th>
<th>Generalization</th>
<th>Interpretability</th>
<th>Stability</th>
</tr>
</thead>
<tbody>
<tr>
<td>Lexicon-based</td>
<td>Low</td>
<td>High (domain-agnostic)</td>
<td>High</td>
<td>High</td>
</tr>
<tr>
<td>Traditional ML</td>
<td>Low&#x2013;Moderate</td>
<td>Moderate</td>
<td>Moderate</td>
<td>High</td>
</tr>
<tr>
<td>Deep Learning</td>
<td>Moderate&#x2013;High</td>
<td>Moderate</td>
<td>Low</td>
<td>High</td>
</tr>
<tr>
<td>Transformers</td>
<td>High</td>
<td>High (fine-tuned)</td>
<td>Low&#x2013;Moderate</td>
<td>High</td>
</tr>
<tr>
<td>LLMs</td>
<td>Very High</td>
<td>Very High (zero-shot)</td>
<td>Low</td>
<td>Low (MVP)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Domain-Specific Challenges</title>
<p>While the methodological advances described in the previous section have driven substantial performance gains on standard benchmarks, real-world deployment of sentiment analysis systems reveals a constellation of domain-specific challenges that remain only partially addressed. This section examines the most pressing of these challenges. <xref ref-type="fig" rid="fig-13">Fig. 13</xref> provides a high-level synthesis of the most commonly reported error categories in sentiment analysis systems, based on recurring patterns observed across benchmark evaluations and qualitative error analyses in the literature.</p>
<fig id="fig-13">
<label>Figure 13</label>
<caption>
<title>Common error types reported in sentiment analysis systems across benchmarks and application domains. Percentages indicate relative prevalence in published error analyses and qualitative studies, reflecting recurring challenges rather than exact empirical frequencies (This figure was generated with the assistance of AI-based tools).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-13.tif"/>
</fig>
<sec id="s5_1">
<label>5.1</label>
<title>Sarcasm and Irony Detection</title>
<p>Sarcasm, where the intended meaning is opposite to the literal meaning, remains one of the most persistent challenges in SA [<xref ref-type="bibr" rid="ref-26">26</xref>,<xref ref-type="bibr" rid="ref-27">27</xref>,<xref ref-type="bibr" rid="ref-124">124</xref>,<xref ref-type="bibr" rid="ref-125">125</xref>]. The utterance &#x201C;Oh great, another meeting&#x201D; conveys negative sentiment despite containing the positive word &#x201C;great,&#x201D; illustrating the fundamental difficulty that sarcasm poses for any system that relies on surface-level lexical cues. Joshi et al. [<xref ref-type="bibr" rid="ref-26">26</xref>] provide a comprehensive survey of automatic sarcasm detection, identifying three broad approaches: rule-based, statistical, and deep learning.</p>
<p>Recent work has specifically evaluated LLM capabilities in this area. Zhang et al. [<xref ref-type="bibr" rid="ref-28">28</xref>] introduced SarcasmBench, the first comprehensive benchmark for evaluating LLMs on sarcasm understanding, and reported that GPT-4 consistently outperformed other tested LLMs across prompting methods, with an average improvement of 14.0%, while all tested LLMs still underperformed supervised PLM-based baselines. In real, contextual understanding of sarcasm requires world knowledge that even large models struggle to apply consistently. Liu et al. [<xref ref-type="bibr" rid="ref-66">66</xref>] proposed CAF-I (Collaborative Agent Framework for Irony Detection), a multi-agent LLM system where specialized agents handle different aspects of irony analysis, literal meaning extraction, context modeling, and incongruity detection. CAF-I achieves state-of-the-art zero-shot irony detection with an average accuracy of 76.89%, suggesting that multi-agent decomposition can mitigate some limitations of single-model irony detection systems. Oprea and B&#x00E2;ra [<xref ref-type="bibr" rid="ref-67">67</xref>] explored an LLM-as-a-judge paradigm for sarcasm detection, using LLMs to evaluate and calibrate the outputs of smaller models; their approach combining DistilBERT-SST2 with LLM-based quality scoring achieved a macro F1 of 0.8784, outperforming both standalone LLMs and standalone fine-tuned models. These results collectively suggest that sarcasm detection benefits most from hybrid architectures that combine the linguistic pattern recognition of fine-tuned models with the pragmatic reasoning capabilities of LLMs.</p>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Domain and Temporal Drift</title>
<p>Sentiment analysis models trained on one domain often perform poorly when applied to another, a phenomenon known as <italic>domain drift</italic> [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-30">30</xref>,<xref ref-type="bibr" rid="ref-126">126</xref>]. Blitzer et al. [<xref ref-type="bibr" rid="ref-29">29</xref>] documented performance drops significantly (often 10&#x2013;15 percentage points) when transferring between books, DVDs, electronics, and kitchen appliance reviews, while Glorot et al. [<xref ref-type="bibr" rid="ref-30">30</xref>] proposed deep learning-based domain adaptation using stacked denoising autoencoders.</p>
<p><italic>Temporal drift</italic> presents a complementary challenge: language evolves over time, and models trained on historical data may fail on contemporary text. Lazaridou et al. [<xref ref-type="bibr" rid="ref-127">127</xref>] showed that language models experience measurable degradation when evaluated on text from time periods outside their training distribution, a finding particularly relevant for social media SA, where slang, memes, and cultural references change rapidly. Sequential adaptation to new data can further exacerbate this issue through catastrophic forgetting, where previously learned knowledge is overwritten by new information [<xref ref-type="bibr" rid="ref-128">128</xref>]. The pre-trained transformer paradigm partially addresses domain drift through transfer learning, as BERT&#x2019;s general linguistic knowledge transfers across domains with minimal fine-tuning [<xref ref-type="bibr" rid="ref-19">19</xref>]. However, LLMs introduce a new form of temporal drift: their training data has a cutoff date, and sentiment about evolving entities (products, public figures, policies) may change post-training.</p>
</sec>
<sec id="s5_3">
<label>5.3</label>
<title>Long-Form Context Processing</title>
<p>Many real-world SA tasks involve long documents, including full movie reviews, earnings call transcripts, and legislative texts, that exceed the context windows of standard models [<xref ref-type="bibr" rid="ref-73">73</xref>,<xref ref-type="bibr" rid="ref-129">129</xref>,<xref ref-type="bibr" rid="ref-130">130</xref>]. BERT&#x2019;s 512-token limit is particularly problematic; while a typical IMDb review contains 230 words on average, the distribution has a long tail extending beyond 2000 words. Beltagy et al. [<xref ref-type="bibr" rid="ref-129">129</xref>] introduced the Longformer with a sparse attention pattern that scales linearly with sequence length, enabling processing of documents up to 4096 tokens. ModernBERT [<xref ref-type="bibr" rid="ref-58">58</xref>] further extends this to 8192 tokens with rotary positional embeddings. LLMs offer the most dramatic improvement, for example Llama 3 supports up to 128K tokens [<xref ref-type="bibr" rid="ref-21">21</xref>]. Recent long-context LLMs such as Gemini 1.5 support context windows approaching one million tokens in certain configurations. While this significantly expands the amount of text that can be processed in a single pass, practical limitations including latency, computational cost, and attention dilution which mean that context length remains a relevant consideration for large-scale sentiment analysis tasks.</p>
</sec>
<sec id="s5_4">
<label>5.4</label>
<title>Cultural and Linguistic Diversity</title>
<p>The vast majority of SA research has focused on English, creating a significant resource gap for other languages [<xref ref-type="bibr" rid="ref-31">31</xref>,<xref ref-type="bibr" rid="ref-32">32</xref>,<xref ref-type="bibr" rid="ref-86">86</xref>,<xref ref-type="bibr" rid="ref-88">88</xref>,<xref ref-type="bibr" rid="ref-131">131</xref>]. Cross-lingual SA aims to transfer sentiment knowledge from high-resource to low-resource languages, with XLM-R [<xref ref-type="bibr" rid="ref-32">32</xref>] serving as a foundational contribution: Conneau et al. trained a multilingual transformer on 100 languages, enabling zero-shot cross-lingual SA by fine-tuning on English data and evaluating on target languages.</p>
<p>Recent advances have expanded this frontier considerably. Miah et al. [<xref ref-type="bibr" rid="ref-61">61</xref>] proposed a multimodal approach to cross-lingual SA across Arabic, Chinese, French, and Italian, using transformer ensembles that combine textual and acoustic features. Zhu et al. [<xref ref-type="bibr" rid="ref-62">62</xref>] introduced a Model Arena framework that systematically compares XLM-R, Llama-3, and GPT-4 for cross-lingual SA, finding that model selection depends critically on the target language family and available supervision. Chen et al. [<xref ref-type="bibr" rid="ref-63">63</xref>] addressed the intersection of cross-lingual and multimodal SA for low-resource languages, introducing the LFD-RT framework that leverages language-family-driven resource transfer. Chen et al. [<xref ref-type="bibr" rid="ref-64">64</xref>] proposed an adaptive self-alignment method for bridging resource gaps in cross-lingual SA, demonstrating improvements for under-resourced language pairs. &#x0160;m&#x00ED;d and Kr&#x00E1;l [<xref ref-type="bibr" rid="ref-33">33</xref>] provide one of the most comprehensive survey to date on cross-lingual aspect-based sentiment analysis, cataloging methods, datasets, and open challenges for this intersection of cross-lingual NLP and ABSA. The LACA framework [<xref ref-type="bibr" rid="ref-65">65</xref>] demonstrates a novel approach using LLM-generated data augmentation for cross-lingual ABSA, where LLMs generate synthetic training data in target languages, partially bridging the annotation gap. Finally, Horsa and Tune [<xref ref-type="bibr" rid="ref-78">78</xref>] and Musa et al. [<xref ref-type="bibr" rid="ref-79">79</xref>] represent important contributions toward SA in African languages (Afaan Oromoo and Hausa, respectively), demonstrating that transformer-based approaches can be adapted for extremely low-resource settings. As shown in <xref ref-type="fig" rid="fig-14">Fig. 14</xref>, cross-lingual transfer remains highly effective for high-resource languages, while performance degrades substantially for low-resource settings, motivating recent African language&#x2013;focused benchmarks and datasets.</p>
<fig id="fig-14">
<label>Figure 14</label>
<caption>
<title>Cross-lingual sentiment analysis performance across language resource levels. Spanish, French, and Chinese values are F1-micro scores from zero-shot cross-lingual transfer (English source) reported in [<xref ref-type="bibr" rid="ref-62">62</xref>], using XLM-R-large (560M, fine-tuned), Llama 3-8B (fine-tuned), and GPT-4 (in-context learning, zero-shot). The Hausa value is the F1-score of HauBERT (mBERT) on supervised ABSA [<xref ref-type="bibr" rid="ref-79">79</xref>], and the Afaan Oromoo value is the accuracy of an SVM with BoW/TF-IDF features on supervised binary ABSA [<xref ref-type="bibr" rid="ref-78">78</xref>]. Illustrative values reflect trends reported across multilingual evaluations, highlighting the <inline-formula id="ieqn-228"><mml:math id="mml-ieqn-228"><mml:mo>&#x223C;</mml:mo></mml:math></inline-formula>30-point degradation from high-resource to low-resource languages and the narrowing gap between fine-tuned and zero-shot approaches in high-resource settings. Metrics and task setups differ across language groups; comparisons are qualitative.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_80601-fig-14.tif"/>
</fig>
<p>Beyond linguistic resource gaps, sentiment analysis faces deeper <bold>cultural nuance challenges</bold> that are not resolved by cross-lingual model transfer alone. Cross-cultural communication patterns can influence how sentiment is expressed linguistically. Some studies suggest that indirect expressions of criticism may appear more frequently in certain cultural contexts, creating additional challenges for sentiment classification across languages and regions. Politeness conventions, humor norms, and the pragmatics of irony vary substantially across language communities, meaning that a model achieving high accuracy on English sarcasm may fail entirely on Japanese <italic>tatemae</italic> (public facade) or Arabic rhetorical understatement. Morphologically rich and agglutinative languages (e.g., Turkish, Finnish, Hausa) present compositional challenges where sentiment modifiers attach as affixes rather than separate words, defeating word-level lexicon approaches and requiring morphology-aware architectures. Code-switching, the mixing of two or more languages within a single utterance, common in multilingual communities creates additional complexity for both tokenization and sentiment inference. These non-linguistic dimensions of cultural diversity remain largely unaddressed in the current benchmarking ecosystem and represent a critical gap for globally deployable SA systems.</p>
</sec>
<sec id="s5_5">
<label>5.5</label>
<title>AI-Generated Fake Reviews</title>
<p>The proliferation of LLMs has created a new and urgent challenge: the generation of sophisticated fake reviews that are increasingly difficult to distinguish from human-written content [<xref ref-type="bibr" rid="ref-37">37</xref>,<xref ref-type="bibr" rid="ref-38">38</xref>]. This represents an adversarial dimension of SA where the same technology used for analysis is weaponized for deception. A multi-domain analysis of LLM-generated fake reviews [<xref ref-type="bibr" rid="ref-37">37</xref>] found that AI-generated open-ended reviews cause a 25%&#x002B; accuracy drop in detection systems compared to template-based fakes, because the increased fluency and coherence of LLM-generated text renders traditional statistical features, n-gram distributions, perplexity scores, far less effective. Research on ChatGPT-paraphrased reviews [<xref ref-type="bibr" rid="ref-38">38</xref>], analyzing patterns across 20 hotels with 6000 fake reviews, revealed that LLM-paraphrased reviews exhibit distinctive patterns in semantic similarity distributions that can be exploited for detection, though these patterns evolve as models improve. The resulting arms race between generation and detection defines this emerging subfield: fine-tuned BERT and RoBERTa classifiers achieve reasonable detection accuracy on current LLM outputs, but generalization to future model versions remains an open problem. Recent regulatory developments reflect growing concern about automated content manipulation. For example, the U.S. Federal Trade Commission&#x2019;s 2024 rule targets deceptive or fabricated consumer reviews and testimonials, including those generated using AI when they misrepresent genuine consumer experiences.</p>
<p>For practitioners deploying SA in review-sensitive domains (e-commerce, hospitality, healthcare), a multi-layer detection pipeline offers the most robust current approach. The first layer employs <italic>statistical features</italic>: perplexity scores from calibrated language models, burstiness metrics (variance in sentence-level surprise), and distributional signatures that differ between human and machine text. The second layer uses <italic>neural classifiers</italic> (fine-tuned RoBERTa or ModernBERT) trained on labeled human/AI review corpora, which capture subtler stylistic differences. The third layer applies <italic>cross-modal verification</italic>: comparing textual sentiment against behavioral signals (purchase history, reviewer profile patterns, temporal posting patterns) to identify reviews that are linguistically plausible but behaviorally anomalous.</p>
<p>A critical production consideration is <italic>detector generalization</italic>: classifiers trained on GPT-3.5-generated text may fail on GPT-4o-generated text, necessitating continual re-training as generative models evolve. This suggests that detection systems must include online monitoring for distributional drift in review characteristics, triggering re-calibration when the detection boundary shifts. The regulatory landscape is also evolving: the U.S. FTC&#x2019;s 2024 rule and emerging EU AI Act provisions create compliance requirements that further motivate investment in detection infrastructure. We emphasize that provably robust defenses against adversarial text generation do not yet exist, making this an active arms race rather than a solved problem.</p>
</sec>
<sec id="s5_6">
<label>5.6</label>
<title>Bias, Fairness, and Ethics</title>
<p>Sentiment analysis systems can perpetuate and amplify societal biases present in training data [<xref ref-type="bibr" rid="ref-132">132</xref>,<xref ref-type="bibr" rid="ref-133">133</xref>]. Kiritchenko and Mohammad [<xref ref-type="bibr" rid="ref-133">133</xref>] evaluated 219 SA systems, finding systematic differences in sentiment scores assigned to sentences mentioning different demographic groups, while Sheng et al. [<xref ref-type="bibr" rid="ref-132">132</xref>] demonstrated that language models generate more negative sentiment text when prompted with certain demographic attributes. These biases have real-world consequences when SA systems are deployed for hiring decisions, content moderation, financial analysis, or public policy evaluation. The intersection of sentiment analysis with online content moderation represents a particularly high-stakes application, where biased sentiment classifiers can systematically over-flag content from certain demographic or linguistic communities, creating disparate impacts on speech and participation. Addressing bias requires both technical interventions (debiasing training data, adversarial training, fairness constraints) and governance frameworks encompassing audit protocols, transparency requirements, and impact assessments.</p>
<p>The challenges outlined above underscore that further progress in sentiment analysis is unlikely to arise from incremental architectural scaling alone. Instead, they point toward the need for fundamentally new system designs that incorporate reasoning, adaptability, privacy preservation, and human oversight. In response, a set of emerging research frontiers has begun to take shape, extending sentiment analysis beyond single-model prediction toward more structured, interactive, and trustworthy frameworks.</p>
</sec>
</sec>
<sec id="s6">
<label>6</label>
<title>Emerging Frontiers</title>
<p>The emerging directions in sentiment analysis, including deployment considerations, explainability, and cross-lingual transfer, are best understood as interconnected challenges rather than isolated research threads. Across these dimensions, a fundamental trade-off persists between performance, interpretability, and stability. High-capacity models such as LLMs offer strong generalization but introduce variability and opacity, while more structured approaches provide interpretability at the cost of flexibility. Similarly, cross-lingual methods extend coverage but often amplify uncertainty due to data scarcity and cultural variation. Understanding sentiment analysis systems through this unified lens highlights that future progress will depend not only on improving accuracy, but on achieving robust, interpretable, and deployable solutions across diverse real-world settings.</p>
<p>The rapid evolution of sentiment analysis, particularly in the era of large language models, has opened several research frontiers that extend well beyond incremental improvements on standard benchmarks. This section surveys the most promising of these directions.</p>
<sec id="s6_1">
<label>6.1</label>
<title>Reasoning-Augmented Sentiment Analysis</title>
<p>Beyond the chain-of-thought approaches discussed in <xref ref-type="sec" rid="s4_5">Section 4.5</xref>, the frontier of reasoning-augmented SA includes multi-step inference systems that decompose complex sentiment tasks into tractable sub-problems. The emergence of reasoning-oriented LLMs suggests a possible direction for sentiment analysis systems that perform more structured multi-step inference when mapping events, context, and expressed opinions to sentiment judgments. Multi-chain CoT for ABSA [<xref ref-type="bibr" rid="ref-57">57</xref>] represents one such advance, employing parallel reasoning chains that independently analyze different aspects before synthesizing a final judgment. Graph-enhanced approaches [<xref ref-type="bibr" rid="ref-117">117</xref>] integrate structured knowledge into the reasoning process, using graph neural networks to model relationships between aspects, opinion expressions, and context.</p>
</sec>
<sec id="s6_2">
<label>6.2</label>
<title>Agentic Sentiment Analysis Workflows</title>
<p>The CAF-I framework for irony detection [<xref ref-type="bibr" rid="ref-66">66</xref>] exemplifies a broader trend toward multi-agent systems for SA. In agentic workflows, specialized LLM agents handle different components of the analysis pipeline: an aspect extraction agent identifies relevant entities and aspects; a context analysis agent models situational and cultural context; a sentiment reasoning agent applies CoT reasoning to determine polarity; and a calibration agent adjusts for known biases and model variability. This decomposition allows each agent to be optimized independently and enables human-in-the-loop oversight at each stage of the pipeline.</p>
<p>The agentic paradigm offers particular advantages for complex SA tasks that involve multiple interacting challenges, for example, cross-lingual sarcasm detection in multimodal content, where a single model would need to simultaneously handle language transfer, pragmatic inference, and modality fusion. By decomposing this into specialized agents, each sub-task can leverage the most appropriate methodology (e.g., a fine-tuned cross-lingual encoder for language transfer, a CoT-prompted LLM for pragmatic inference, and a dedicated fusion module for multimodal integration). The orchestration of these agents, including conflict resolution when agents disagree and confidence calibration across the pipeline, represents an active research frontier with connections to the broader multi-agent systems literature.</p>
</sec>
<sec id="s6_3">
<label>6.3</label>
<title>Federated and Privacy-Preserving Sentiment Analysis</title>
<p>Healthcare and financial sentiment analysis (SA) applications often require processing sensitive data under strict privacy constraints. Federated learning (FL) enables training SA models across distributed datasets without centralizing raw data, by iteratively aggregating locally computed updates from multiple clients.</p>
<p>Formally, let <italic>K</italic> denote the number of participating clients and <inline-formula id="ieqn-229"><mml:math id="mml-ieqn-229"><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula> the global model parameters at communication round <inline-formula id="ieqn-230"><mml:math id="mml-ieqn-230"><mml:mi>t</mml:mi></mml:math></inline-formula>. The global model is updated as:<disp-formula id="eqn-32"><label>(32)</label><mml:math id="mml-eqn-32" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>K</mml:mi></mml:mfrac><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msubsup><mml:mi>&#x03B8;</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>here, <inline-formula id="ieqn-231"><mml:math id="mml-ieqn-231"><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msubsup><mml:mi>&#x03B8;</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> represents the update computed by client <inline-formula id="ieqn-232"><mml:math id="mml-ieqn-232"><mml:mi>k</mml:mi></mml:math></inline-formula> after performing local optimization on its private dataset. In practice, these updates are typically obtained via one or more steps of stochastic gradient descent (SGD), i.e.,
<disp-formula id="ueqn-33"><mml:math id="mml-ueqn-33" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msubsup><mml:mi>&#x03B8;</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup><mml:mo>&#x2248;</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:msub><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-233"><mml:math id="mml-ieqn-233"><mml:msub><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msub></mml:math></inline-formula> denotes the local loss function and <inline-formula id="ieqn-234"><mml:math id="mml-ieqn-234"><mml:mi>&#x03B7;</mml:mi></mml:math></inline-formula> is the learning rate. The aggregation step in <xref ref-type="disp-formula" rid="eqn-32">Eq. (32)</xref> approximates a descent direction for the global objective <inline-formula id="ieqn-235"><mml:math id="mml-ieqn-235"><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mi>k</mml:mi></mml:msub></mml:math></inline-formula> without requiring data centralization.</p>
<p>The formulation above assumes equal contribution from all clients. In practice, federated averaging typically uses a weighted aggregation based on local dataset sizes:<disp-formula id="ueqn-34"><mml:math id="mml-ueqn-34" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover><mml:mfrac><mml:msub><mml:mi>n</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mi>n</mml:mi></mml:mfrac><mml:msubsup><mml:mi>&#x03B8;</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-236"><mml:math id="mml-ieqn-236"><mml:msub><mml:mi>n</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:math></inline-formula> is the number of samples at client <inline-formula id="ieqn-237"><mml:math id="mml-ieqn-237"><mml:mi>k</mml:mi></mml:math></inline-formula>, <inline-formula id="ieqn-238"><mml:math id="mml-ieqn-238"><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mi>n</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:math></inline-formula>, and <inline-formula id="ieqn-239"><mml:math id="mml-ieqn-239"><mml:msubsup><mml:mi>&#x03B8;</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msup><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msubsup><mml:mi>&#x03B8;</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> denotes the locally updated model. To further enhance privacy, differential privacy (DP) can be incorporated by perturbing client updates before aggregation:<disp-formula id="eqn-33"><label>(33)</label><mml:math id="mml-eqn-33" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msub><mml:mrow><mml:mover><mml:mi>&#x03B8;</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>k</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mrow><mml:mi>&#x1D4A9;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mi>I</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>In practice, DP is implemented by first clipping updates to a fixed norm bound <italic>C</italic>,
<disp-formula id="ueqn-36"><mml:math id="mml-ueqn-36" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">&#x2190;</mml:mo><mml:mfrac><mml:mrow><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:msub><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>C</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>followed by the addition of Gaussian noise. The noise term <inline-formula id="ieqn-240"><mml:math id="mml-ieqn-240"><mml:mrow><mml:mi>&#x1D4A9;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msup><mml:mi>&#x03C3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mi>I</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is calibrated to the sensitivity of the clipped updates, where <inline-formula id="ieqn-241"><mml:math id="mml-ieqn-241"><mml:mi>&#x03C3;</mml:mi></mml:math></inline-formula> controls the privacy&#x2013;utility trade-off: larger <inline-formula id="ieqn-242"><mml:math id="mml-ieqn-242"><mml:mi>&#x03C3;</mml:mi></mml:math></inline-formula> provides stronger privacy guarantees at the cost of reduced model accuracy. Intuitively, the injected noise obscures the contribution of individual data points, making it difficult to infer sensitive information from shared updates.</p>
<p>In practice, federated SA introduces additional challenges beyond standard FL settings. Sentiment label distributions are often highly non-i.i.d. across clients (e.g., different institutions may exhibit distinct linguistic patterns or domain-specific sentiment cues), requiring robust aggregation strategies such as FedProx. Moreover, the privacy&#x2013;utility trade-off is particularly pronounced for SA, where high-dimensional transformer representations lead to large gradient magnitudes, necessitating stronger noise injection for DP guarantees and potentially degrading fine-grained sentiment distinctions.</p>
</sec>
<sec id="s6_4">
<label>6.4</label>
<title>Real-Time and Edge Deployment</title>
<p>Deploying SA at scale requires efficient models, and two key techniques from the LLM efficiency literature are particularly relevant. LoRA (Low-Rank Adaptation) [<xref ref-type="bibr" rid="ref-134">134</xref>] freezes the pre-trained weights and injects trainable low-rank decomposition matrices:<disp-formula id="eqn-34"><label>(34)</label><mml:math id="mml-eqn-34" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msup><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mn>0</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mn>0</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="bold">B</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="bold">A</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>here, <inline-formula id="ieqn-243"><mml:math id="mml-ieqn-243"><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mn>0</mml:mn></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> denotes the frozen pre-trained weight matrix, and the low-rank update <inline-formula id="ieqn-244"><mml:math id="mml-ieqn-244"><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="bold">B</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="bold">A</mml:mi></mml:mrow></mml:math></inline-formula> has rank at most <inline-formula id="ieqn-245"><mml:math id="mml-ieqn-245"><mml:mi>r</mml:mi></mml:math></inline-formula>, ensuring that <inline-formula id="ieqn-246"><mml:math id="mml-ieqn-246"><mml:msup><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mo>&#x2032;</mml:mo></mml:msup></mml:math></inline-formula> has the same dimensionality as <inline-formula id="ieqn-247"><mml:math id="mml-ieqn-247"><mml:msub><mml:mrow><mml:mi mathvariant="bold">W</mml:mi></mml:mrow><mml:mn>0</mml:mn></mml:msub></mml:math></inline-formula>. <inline-formula id="ieqn-248"><mml:math id="mml-ieqn-248"><mml:mrow><mml:mi mathvariant="bold">B</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, <inline-formula id="ieqn-249"><mml:math id="mml-ieqn-249"><mml:mrow><mml:mi mathvariant="bold">A</mml:mi></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, and <inline-formula id="ieqn-250"><mml:math id="mml-ieqn-250"><mml:mi>r</mml:mi><mml:mo>&#x226A;</mml:mo><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>d</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>, reducing trainable parameters by over two orders of magnitude while maintaining competitive performance. QLoRA [<xref ref-type="bibr" rid="ref-135">135</xref>] extends extends this with 4-bit quantization, enabling memory-efficient fine-tuning of very large models and making local or resource-constrained deployment substantially more feasible than full-precision alternatives.</p>
<p>Deployment considerations vary substantially with hardware, quantization level, batching strategy, sequence length, and serving framework. In general, lexicon-based systems are the most lightweight and latency-efficient; fine-tuned encoder models such as BERT-base are typically practical for high-throughput production inference; quantized mid-scale LLMs can support local or on-premise deployment with substantially higher latency and memory demands; and frontier-scale LLMs usually require either hosted APIs or multi-accelerator infrastructure. Accordingly, deployment comparisons should be reported together with the underlying hardware and inference configuration rather than as hardware-independent universal numbers.</p>
</sec>
<sec id="s6_5">
<label>6.5</label>
<title>Explainable Sentiment Analysis</title>
<p>As SA systems are deployed in high-stakes settings, explainability becomes critical. Danilevsky et al. [<xref ref-type="bibr" rid="ref-136">136</xref>] survey XAI for NLP, identifying attention visualization, feature attribution, and natural language explanations as primary explanation modalities. Two widely adopted frameworks deserve particular attention.</p>
<p>LIME (Local Interpretable Model-agnostic Explanations) [<xref ref-type="bibr" rid="ref-137">137</xref>] explains individual predictions by fitting a local linear model:<disp-formula id="eqn-35"><label>(35)</label><mml:math id="mml-eqn-35" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>&#x03BE;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>arg</mml:mi><mml:mo>&#x2061;</mml:mo><mml:munder><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mrow><mml:mi>g</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mi>&#x1D4A2;</mml:mi></mml:mrow></mml:mrow></mml:munder><mml:mtext>&#xA0;</mml:mtext><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>f</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi mathvariant="normal">&#x03A9;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-251"><mml:math id="mml-ieqn-251"><mml:mi>f</mml:mi></mml:math></inline-formula> is the black-box model, <inline-formula id="ieqn-252"><mml:math id="mml-ieqn-252"><mml:mi>g</mml:mi></mml:math></inline-formula> is an interpretable surrogate, <inline-formula id="ieqn-253"><mml:math id="mml-ieqn-253"><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula> defines a locality kernel, and <inline-formula id="ieqn-254"><mml:math id="mml-ieqn-254"><mml:mi mathvariant="normal">&#x03A9;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> penalizes complexity. <inline-formula id="ieqn-255"><mml:math id="mml-ieqn-255"><mml:mrow><mml:mi>&#x02112;</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>f</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> measures the fidelity of the surrogate model <inline-formula id="ieqn-256"><mml:math id="mml-ieqn-256"><mml:mi>g</mml:mi></mml:math></inline-formula> in approximating the black-box model <inline-formula id="ieqn-257"><mml:math id="mml-ieqn-257"><mml:mi>f</mml:mi></mml:math></inline-formula> in the locality defined by <inline-formula id="ieqn-258"><mml:math id="mml-ieqn-258"><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula>, and <inline-formula id="ieqn-259"><mml:math id="mml-ieqn-259"><mml:mrow><mml:mi>&#x1D4A2;</mml:mi></mml:mrow></mml:math></inline-formula> denotes the class of interpretable models. SHAP (SHapley Additive exPlanations) provides theoretically grounded feature attributions based on cooperative game theory, computing the SHAP value for feature <inline-formula id="ieqn-260"><mml:math id="mml-ieqn-260"><mml:mi>j</mml:mi></mml:math></inline-formula> as:<disp-formula id="eqn-36"><label>(36)</label><mml:math id="mml-eqn-36" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:msub><mml:mi>&#x03D5;</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:munder><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mo>&#x2286;</mml:mo><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow><mml:mo>\</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mi>j</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo></mml:mrow></mml:munder><mml:mfrac><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>S</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>!</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mtext mathvariant="bold">F</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>S</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>!</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>!</mml:mo></mml:mrow></mml:mfrac><mml:mrow><mml:mo>[</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>S</mml:mi><mml:mo>&#x222A;</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mi>j</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>S</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-261"><mml:math id="mml-ieqn-261"><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow></mml:math></inline-formula> is the full feature set and <italic>S</italic> ranges over all subsets excluding feature <inline-formula id="ieqn-262"><mml:math id="mml-ieqn-262"><mml:mi>j</mml:mi></mml:math></inline-formula>. <inline-formula id="ieqn-263"><mml:math id="mml-ieqn-263"><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>S</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> denotes the model output when only the features in subset <italic>S</italic> are present (with other features marginalized or fixed to baseline values), and <inline-formula id="ieqn-264"><mml:math id="mml-ieqn-264"><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow></mml:math></inline-formula> is the full set of input features.</p>
<p>Recent applications of XAI to SA have been diverse and promising. ModernBERT has been combined with SHAP and LIME for transparent sentiment classification [<xref ref-type="bibr" rid="ref-68">68</xref>], suggesting that modern encoder architectures can support both strong predictive performance and post hoc interpretability. Explainable ABSA using transformers with LIME, SHAP, attention visualization, integrated gradients, and Grad-CAM [<xref ref-type="bibr" rid="ref-69">69</xref>] provides a comprehensive toolkit to understand aspect-level predictions. In a domain-specific application, ABSA for software requirements elicitation with XAI [<xref ref-type="bibr" rid="ref-70">70</xref>] achieved F1 &#x003D; 0.94 on the ACP dataset by combining BERT with LIME explanations, enabling software engineers to understand and trust aspect-level sentiment predictions. Rajani et al. [<xref ref-type="bibr" rid="ref-138">138</xref>] reported that generating natural language explanations alongside predictions can improve both model performance and user trust. This is increasingly relevant for LLM-based sentiment analysis, although fluency of explanation should not be conflated with explanation faithfulness or causal correctness.</p>
<p>To complement the largely conceptual discussion of explainability and reasoning mechanisms, <xref ref-type="table" rid="table-5">Table 5</xref> summarizes representative quantitative findings from prior studies, highlighting the measurable impact of prompt design, reasoning strategies, and model choice on sentiment analysis performance.</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Reported performance impact of explainability and prompt variability in sentiment analysis models. Values are representative ranges reported in prior studies with heterogeneous experimental conditions (different datasets, baselines, and LLM versions); direct quantitative comparison across rows is not intended.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th>Factor</th>
<th>Observed Impact</th>
<th>Reference Insight</th>
</tr>
</thead>
<tbody>
<tr>
<td>Prompt variation (LLMs)</td>
<td>10%&#x2013;15% accuracy variation</td>
<td>SentiEval benchmark [<xref ref-type="bibr" rid="ref-22">22</xref>]</td>
</tr>
<tr>
<td>LLM vs. Lexicon correlation</td>
<td>r &#x003D; 0.59&#x2013;0.77 vs. 0.20&#x2013;0.30</td>
<td>Rathje et al. [<xref ref-type="bibr" rid="ref-24">24</xref>]</td>
</tr>
<tr>
<td>CoT reasoning (implicit SA)</td>
<td>&#x002B;5%&#x2013;12% improvement</td>
<td>THOR [<xref ref-type="bibr" rid="ref-54">54</xref>]/SAoT [<xref ref-type="bibr" rid="ref-55">55</xref>] studies</td>
</tr>
<tr>
<td>CoT on simple tasks</td>
<td>&#x2212;2%&#x2013;5% degradation</td>
<td>Zheng et al. [<xref ref-type="bibr" rid="ref-56">56</xref>]</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We emphasize that the ranges summarized in <xref ref-type="table" rid="table-5">Table 5</xref> are drawn from studies with heterogeneous experimental setups i.e., different datasets, baselines, prompt formats, and LLM versions and the source papers do not consistently report normalized variance, confidence intervals, or significance tests for the quoted deltas. Accordingly, the reported magnitudes should be read as indicative order-of-effect estimates rather than as directly comparable measurements, and side-by-side numerical comparison between rows is not intended.</p>
<p>As shown in <xref ref-type="table" rid="table-5">Table 5</xref>, explainability-related factors such as prompt formulation and reasoning strategies are associated with non-trivial variability. These results reinforce that explainability is not merely an interpretive layer, but an active component influencing model behavior, particularly in LLM-based systems.</p>
<p>We further caution that the associations in <xref ref-type="table" rid="table-5">Table 5</xref> are largely correlational: most source studies vary prompts or reasoning strategies without holding model version, decoding temperature, in-context examples, and dataset composition strictly constant, so the observed deltas conflate prompt effects with model and data effects. Controlled ablations that perturb only the prompt or reasoning scaffold while fixing the underlying model and decoding configuration remain scarce, and are a prerequisite for treating prompt engineering or chain-of-thought as reliable, isolable performance levers rather than joint artifacts of a particular deployment configuration.</p>
</sec>
<sec id="s6_6">
<label>6.6</label>
<title>Cross-Lingual Transfer and Low-Resource Languages</title>
<p>The cross-lingual frontier extends beyond simple model transfer to address deeper structural challenges. &#x0160;m&#x00ED;d and Kr&#x00E1;l&#x2019;s comprehensive survey [<xref ref-type="bibr" rid="ref-33">33</xref>] identifies key open problems including aspect term extraction across languages with different morphological structures, sentiment compositionality in agglutinative languages, and evaluation standardization across language pairs. The LACA framework [<xref ref-type="bibr" rid="ref-65">65</xref>] represents a novel LLM-augmented approach where large language models generate synthetic aspect-based sentiment training data in target languages, effectively using the LLM&#x2019;s multilingual knowledge as a data augmentation engine. Chen et al. [<xref ref-type="bibr" rid="ref-63">63</xref>] further advance this direction with language-family-driven resource transfer, exploiting linguistic typological similarities between related languages. Chen et al. [<xref ref-type="bibr" rid="ref-64">64</xref>] introduce adaptive self-alignment for bridging resource gaps, where models learn to align sentiment representations across languages without parallel corpora, using self-supervised objectives on monolingual data.</p>
<p><bold>Critical Assessment of Synthetic Data Approaches:</bold> While frameworks such as LACA offer pragmatic solutions for data-scarce settings, they also introduce important risks. When downstream models are trained heavily on LLM-generated sentiment data, several failure modes may arise, including reduced output diversity, amplification of biases present in the generating model, and erosion of subtle human-annotated distinctions. These concerns do not invalidate synthetic data augmentation, but they make rigorous auditing against human-annotated data essential. First, <italic>model collapse</italic>: iterative training on LLM-generated data can cause progressive narrowing of output distributions, as each generation cycle reinforces the generating model&#x2019;s biases and reduces diversity. Second, <italic>bias amplification</italic>: systematic biases in the generating LLM such as sentiment skew toward certain topics, cultural biases inherited from English-centric pre-training, and stylistic homogeneity are inherited and potentially magnified by the downstream model. Third, <italic>erosion of ground truth</italic>: subtle human-annotated nuances, ambivalent sentiment, culture-specific pragmatic markers, implicit opinions expressed through discourse-level patterns may be systematically smoothed out by LLM-generated data that favors prototypical, unambiguous examples. These risks do not invalidate synthetic data approaches, but they demand rigorous quality auditing: distributional comparison between synthetic and human-annotated data, human validation of representative synthetic samples, and evaluation on held-out human-annotated test sets to ensure that synthetic training does not degrade performance on genuinely ambiguous or culturally nuanced inputs.</p>
<p>Taken together, the empirical patterns summarized in <xref ref-type="table" rid="table-5">Table 5</xref> and the methodological limitations discussed throughout <xref ref-type="sec" rid="s6">Section 6</xref> map directly onto the research agenda developed in <xref ref-type="sec" rid="s7">Section 7</xref>. The <inline-formula id="ieqn-265"><mml:math id="mml-ieqn-265"><mml:mn>10</mml:mn></mml:math></inline-formula>%&#x2013;<inline-formula id="ieqn-266"><mml:math id="mml-ieqn-266"><mml:mn>15</mml:mn><mml:mi mathvariant="normal">&#x0025;</mml:mi></mml:math></inline-formula> accuracy swings across prompt formulations reported on SentiEval [<xref ref-type="bibr" rid="ref-22">22</xref>] reflect the absence of standardized, consistency-aware evaluation protocols for LLM-based SA, motivating the multi-axis evaluation framework and mandatory CIR reporting proposed in <xref ref-type="sec" rid="s7_1">Sections 7.1</xref> and <xref ref-type="sec" rid="s7_2">7.2</xref>. The asymmetric effects of chain-of-thought: <inline-formula id="ieqn-267"><mml:math id="mml-ieqn-267"><mml:mo>+</mml:mo><mml:mn>5</mml:mn></mml:math></inline-formula>%&#x2013;<inline-formula id="ieqn-268"><mml:math id="mml-ieqn-268"><mml:mn>12</mml:mn><mml:mi mathvariant="normal">&#x0025;</mml:mi></mml:math></inline-formula> on implicit sentiment tasks [<xref ref-type="bibr" rid="ref-54">54</xref>,<xref ref-type="bibr" rid="ref-55">55</xref>] but a <inline-formula id="ieqn-269"><mml:math id="mml-ieqn-269"><mml:mn>2</mml:mn></mml:math></inline-formula>%&#x2013;<inline-formula id="ieqn-270"><mml:math id="mml-ieqn-270"><mml:mn>5</mml:mn><mml:mi mathvariant="normal">&#x0025;</mml:mi></mml:math></inline-formula> degradation on straightforward polarity classification [<xref ref-type="bibr" rid="ref-56">56</xref>] expose the lack of controlled ablations in current LLM-SA research and underpin the call in <xref ref-type="sec" rid="s7_3">Section 7.3</xref> for human&#x2013;AI collaborative annotation and standardized experimental reporting (exact prompts, temperature, seeds, and run counts). The large gap between LLM and lexicon correlations with human annotations (<inline-formula id="ieqn-271"><mml:math id="mml-ieqn-271"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.59</mml:mn></mml:math></inline-formula>&#x2013;<inline-formula id="ieqn-272"><mml:math id="mml-ieqn-272"><mml:mn>0.77</mml:mn></mml:math></inline-formula> vs. <inline-formula id="ieqn-273"><mml:math id="mml-ieqn-273"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.20</mml:mn></mml:math></inline-formula>&#x2013;<inline-formula id="ieqn-274"><mml:math id="mml-ieqn-274"><mml:mn>0.30</mml:mn></mml:math></inline-formula> [<xref ref-type="bibr" rid="ref-24">24</xref>]) is a performance signal, not an explanation; understanding <italic>why</italic> LLMs track human judgment better, and under which linguistic and cultural conditions they fail to do so, requires the causally grounded, cognitively informed SA proposed in <xref ref-type="sec" rid="s7_4">Section 7.4</xref>. In this sense, the patterns in <xref ref-type="table" rid="table-5">Table 5</xref> are not isolated empirical curiosities but diagnostic symptoms of deeper methodological gaps: evaluation instability, uncontrolled confounding, and absent causal grounding that the next section seeks to make explicit and addressable.</p>
</sec>
</sec>
<sec id="s7">
<label>7</label>
<title>Research Gaps and Future Directions</title>
<p>Despite the remarkable progress chronicled in this survey, the field of sentiment analysis stands at an inflection point where the most impactful advances are likely to come not from incremental refinements of existing methods, but from rethinking the fundamental assumptions, evaluation paradigms, and application contexts of the discipline. This section identifies key research gaps and articulates a forward-looking vision for the next generation of sentiment analysis systems. To provide actionable guidance, we organize future directions into three priority tiers: <italic>Tier 1 (Immediate, 1&#x2013;2 years)</italic>, addressing gaps where solutions are feasible with current technology; <italic>Tier 2 (Medium-term, 2&#x2013;5 years)</italic>, requiring significant research but with clear pathways; and <italic>Tier 3 (Long-term, 5&#x002B; years)</italic>, representing aspirational goals that require fundamental advances.</p>
<sec id="s7_1">
<label>7.1</label>
<title>Integrated Evaluation Frameworks (Tier 1)</title>
<p>Despite the proliferation of benchmarks, the SA field lacks a unified evaluation framework that captures the full spectrum of capabilities. SentiEval [<xref ref-type="bibr" rid="ref-22">22</xref>] represents the most comprehensive effort to date, but it focuses primarily on English text and does not cover multimodal or cross-lingual settings. A truly comprehensive evaluation framework would need to jointly assess performance across granularity levels (document, sentence, aspect), robustness to domain and temporal drift, cross-lingual generalization, multimodal integration quality, computational efficiency and latency, explanation quality and faithfulness, and consistency across runs to address the MVP [<xref ref-type="bibr" rid="ref-25">25</xref>]. We envision such a framework as a &#x201C;living benchmark&#x201D; that evolves with the field, continuously incorporating new domains, languages, and modalities as they become relevant. Such a framework would need to move beyond static leaderboards toward dynamic evaluations that test models under distribution shift, adversarial conditions, and low-resource constraints simultaneously, thereby providing a holistic view of model readiness for real-world deployment.</p>
<p><bold>Proposed Evaluation Protocol:</bold> As a concrete first step, we recommend a multi-axis evaluation protocol that the community could adopt: (1) mandatory cross-domain evaluation, with training on one domain and testing on at least two held-out domains; (2) temporal robustness testing, with evaluation on data from time periods outside the training distribution; (3) demographic fairness auditing using the methodology of Kiritchenko and Mohammad [<xref ref-type="bibr" rid="ref-133">133</xref>], reporting sentiment score disparities across demographic mentions; (4) consistency testing across at least 5 inference runs for any stochastic model, reporting CIR alongside accuracy; and (5) multilingual evaluation covering at least one high-resource, one medium-resource, and one low-resource language. This protocol would enable meaningful cross-study comparisons and expose robustness gaps hidden by standard single-dataset evaluation.</p>
</sec>
<sec id="s7_2">
<label>7.2</label>
<title>Standardization Challenges (Tier 1)</title>
<p>The field suffers from inconsistent experimental practices: different papers use different data splits, preprocessing pipelines, and evaluation metrics, making direct comparison difficult. The green AI movement [<xref ref-type="bibr" rid="ref-123">123</xref>] further advocates for standardized reporting of computational costs alongside accuracy metrics, recognizing that the environmental and economic costs of training and deploying ever-larger models must be weighed against their marginal performance gains. Looking ahead, we believe the community should converge on standardized evaluation protocols that include mandatory reporting of carbon footprint, inference latency, and memory requirements alongside accuracy. Such protocols would accelerate progress by making results more comparable and would promote the development of models that achieve the best trade-off between performance and resource consumption, thereby democratizing access to high-quality SA across institutions with varying computational budgets.</p>
</sec>
<sec id="s7_3">
<label>7.3</label>
<title>Human-AI Collaborative Annotation (Tier 2)</title>
<p>The comparison of LLMs with human annotators [<xref ref-type="bibr" rid="ref-53">53</xref>] reveals nuanced complementarities: humans excel at context-dependent judgments requiring cultural knowledge, while LLMs provide superior consistency and throughput. Future annotation pipelines will likely combine LLM pre-annotation with human validation, requiring new quality control protocols and annotation interfaces. Yin et al. [<xref ref-type="bibr" rid="ref-139">139</xref>] established benchmarking protocols for zero-shot text classification that can serve as a template for standardized SA evaluation, though adaptation to the specific characteristics of sentiment tasks (subjectivity, granularity, domain dependence) remains necessary. We anticipate that the most transformative development in this space will be the design of interactive annotation systems in which humans and AI models work in tight feedback loops, each correcting and calibrating the other. Such systems would not only produce higher-quality labeled datasets but also yield insights into the systematic differences between human and machine understanding of sentiment, informing the design of more robust and culturally sensitive models. As an immediate action item, we recommend that all SA publications report: (a) exact data splits and preprocessing steps sufficient for reproduction; (b) inference latency and memory footprint; (c) carbon footprint estimates (using tools such as CodeCarbon or ML CO<sub>2</sub> Impact); and (d) for LLM-based methods, prompt text, temperature setting, and number of inference runs.</p>
</sec>
<sec id="s7_4">
<label>7.4</label>
<title>Toward Causal and Cognitively Grounded Sentiment Analysis (Tier 3)</title>
<p>Perhaps the most consequential gap in the current literature is the absence of <italic>causal</italic> sentiment analysis. Existing methods, from lexicons to LLMs, are fundamentally correlational: they identify associations between textual features and sentiment labels but cannot explain <italic>why</italic> a text conveys a particular sentiment or predict <italic>what would change the sentiment</italic>. A causal approach would enable counterfactual reasoning (e.g., &#x201C;If the product had arrived on time, would the review be positive?&#x201D;), opening new applications in customer experience management, policy analysis, and product design. Achieving causal SA will require integration with causal inference frameworks and may benefit from cognitive science insights into how humans form and update opinions. Concretely, this means developing models that can distinguish between sentiment triggers (the causal antecedents of an expressed opinion) and sentiment indicators (the linguistic markers that happen to correlate with polarity), a distinction that current approaches largely ignore. This connects to the sarcasm detection challenge (<xref ref-type="sec" rid="s5_1">Section 5.1</xref>): understanding sarcasm fundamentally requires causal reasoning about the speaker&#x2019;s communicative intent, linking the pragmatic inference techniques developed for figurative language to the broader goal of causal SA.</p>
</sec>
<sec id="s7_5">
<label>7.5</label>
<title>Dynamic Sentiment Tracking and Temporal Intelligence (Tier 2)</title>
<p>Current sentiment analysis operates predominantly in a static, snapshot mode: a model processes a fixed piece of text and produces a classification. The real world, however, is dynamic. Public opinion about a product, policy, or public figure shifts over time in response to events, and these shifts often follow predictable patterns of escalation, equilibrium, and decay. The next generation of SA systems should be capable of <italic>dynamic sentiment tracking</italic>, monitoring evolving sentiment in streaming data through online learning and drift detection mechanisms. Such systems would need to combine temporal modeling techniques with SA, detecting not only what sentiment is expressed but when and why it changes. This vision extends to <italic>predictive</italic> sentiment analysis, in which models forecast future sentiment trajectories based on current trends and external events, enabling proactive rather than reactive decision-making in domains such as brand management, financial markets, and public health surveillance. This direction intersects with domain drift (<xref ref-type="sec" rid="s5_2">Section 5.2</xref>): the temporal evolution of sentiment about a specific entity involves both the genuine shift in public opinion and the linguistic drift that changes how opinions are expressed, requiring models that can disentangle these two sources of temporal variation.</p>
</sec>
<sec id="s7_6">
<label>7.6</label>
<title>Sentiment Grounding in Real-World Outcomes (Tier 3)</title>
<p>A related and equally important gap is <italic>sentiment grounding</italic>: connecting textual sentiment to tangible real-world outcomes such as stock prices, election results, product sales, and public health metrics in a principled and validated manner. While correlations between sentiment signals and market movements have been widely reported, rigorous causal validation remains rare. Future work should establish when and under what conditions aggregated sentiment signals can serve as reliable leading indicators, and should develop calibrated uncertainty estimates for sentiment-based predictions. This requires moving beyond retrospective analyses to prospective, pre-registered studies that test sentiment-based forecasts against ground-truth outcomes. The multimodal fusion techniques from <xref ref-type="sec" rid="s4_6">Section 4.6</xref> could benefit healthcare applications where patient sentiment expressed through both text and vocal cues needs to be grounded in clinical outcomes, representing a cross-domain linkage between methodological advances and practical deployment.</p>
</sec>
<sec id="s7_7">
<label>7.7</label>
<title>Adversarial Robustness and Trust (Tier 2)</title>
<p>The emergence of AI-generated fake reviews [<xref ref-type="bibr" rid="ref-37">37</xref>,<xref ref-type="bibr" rid="ref-38">38</xref>] is a harbinger of a broader challenge: defending SA systems against deliberate manipulation. As LLMs become more capable, the sophistication of adversarial attacks on sentiment systems will increase correspondingly, potentially encompassing not only fake reviews but also coordinated influence campaigns, synthetic social media personas, and targeted manipulation of financial sentiment indicators. Building adversarially robust SA systems will require a combination of provable defense mechanisms, continuous monitoring for distributional anomalies, and cross-modal verification strategies that triangulate sentiment signals across text, behavior, and contextual metadata. The broader goal is to develop SA systems that are not merely accurate in benign settings but demonstrably <italic>trustworthy</italic> in adversarial ones, a property that will be essential for deployment in critical infrastructure such as financial markets, healthcare systems, and democratic institutions. This challenge connects to the bias and fairness concerns of <xref ref-type="sec" rid="s5_6">Section 5.6</xref>: both adversarial robustness and fairness involve distribution shift between training and deployment conditions, and techniques developed for one (e.g., distributionally robust optimization) may benefit the other.</p>
</sec>
<sec id="s7_8">
<label>7.8</label>
<title>Toward a Unified Vision</title>
<p>The convergence of several trends, including reasoning-augmented models, explainability frameworks, federated learning, and agentic architectures, points toward a future where SA systems are not merely accurate but also reliable, interpretable, fair, and privacy-preserving. Achieving this vision requires interdisciplinary collaboration spanning NLP, human-computer interaction, ethics, cognitive science, and domain expertise. The ultimate aspiration is a new class of sentiment intelligence systems that understand opinions with the depth and nuance of a human analyst while operating with the scale, speed, and consistency that only machines can provide: systems that can be trusted precisely because they can explain their reasoning, acknowledge their uncertainty, and adapt to the evolving landscape of human expression.</p>
</sec>
</sec>
<sec id="s8">
<label>8</label>
<title>Conclusion</title>
<p>This survey has traced the evolution of sentiment analysis from its origins in lexicon-based methods through classical machine learning, deep learning, pre-trained transformers, and the current era of large language models. Our analysis reveals that sentiment analysis has matured from a binary classification task on movie reviews into a multi-dimensional research area spanning multiple granularity levels, modalities, languages, and domains.</p>
<p>The current state of the field is characterized by a productive tension between two paradigms: fine-tuned small language models (BERT, RoBERTa, ModernBERT) that achieve high accuracy with efficient inference, and large language models (GPT-4, Llama 3) that offer remarkable zero-shot flexibility at greater computational cost. The SentiEval evaluation demonstrates that neither paradigm uniformly dominates; fine-tuned SLMs lead on standard benchmarks while LLMs excel in few-shot and cross-domain settings. The model variability problem adds urgency to the development of reliable evaluation protocols that can account for the inherent stochasticity of LLM-based approaches.</p>
<p>Emerging frontiers, including chain-of-thought reasoning for implicit sentiment, multimodal LLMs, agentic workflows, cross-lingual transfer, explainable SA, and AI-generated fake review detection, collectively define a research agenda that extends well beyond incremental benchmark improvements. Addressing these frontiers will require not only algorithmic innovation but also new datasets, evaluation frameworks, and interdisciplinary collaboration between NLP researchers, domain experts, ethicist, and practitioners.</p>
<p>Sentiment analysis remains, after more than two decades, both a problem of enduring practical importance and a persistent scientific challenge. While the field has progressed from hand-crafted lexicons to trillion-parameter language models, this evolution has not eliminated the fundamental difficulty at its core. Accurately inferring sentiment ultimately requires understanding intent, context, and nuance&#x2014;what people mean rather than merely what they say. Addressing this gap will define the next phase of sentiment analysis research and determine its reliability in real-world applications.</p>
</sec>
</body>
<back>
<ack>
<p>The authors acknowledge the use of AI model, OpenAI GPT 5.2 for generating <xref ref-type="fig" rid="fig-1">Figs. 1</xref>&#x2013;<xref ref-type="fig" rid="fig-5">5</xref> and <xref ref-type="fig" rid="fig-9">9</xref>&#x2013;<xref ref-type="fig" rid="fig-13">13</xref> as well as for improving the language quality and presentation of the manuscript. The authors take full responsibility for the accuracy, integrity, and originality of the content.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>This project was funded by the Deanship of Scientific Research (DSR) at King Abdulaziz University, Jeddah, Saudi Arabia under grant no. (IPP: 543-305-2025). The authors, therefore, acknowledge with thanks DSR for technical and financial support.</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>Conceptualization, methodology, literature review, data curation, visualization, and writing&#x2014;original draft preparation were carried out by Shuvodeep De, Agnivo Gosai, and Karun Thankachan. Visualization, resources, and writing&#x2014;review and editing were contributed by Ramadan A. ZeinEldin and Abdulaziz T. Almaktoom. Supervision and writing&#x2014;review and editing were provided by Mustafa Bayram and Ali Wagdy Mohamed. Project administration was performed by Ali Wagdy Mohamed. All authors reviewed and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>No new data were created or analyzed in this study. All data supporting the findings of this work are derived from previously published studies, which have been appropriately cited in the manuscript.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not Applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest.</p>
</sec>
<glossary content-type="abbreviations" id="glossary-1">
<title>Abbreviations</title>
<p>The following abbreviations are used in this manuscript</p>
<def-list>
<def-item>
<term>ABSA</term>
<def>
<p>Aspect-Based Sentiment Analysis</p>
</def>
</def-item>
<def-item>
<term>AI</term>
<def>
<p>Artificial Intelligence</p>
</def>
</def-item>
<def-item>
<term>API</term>
<def>
<p>Application Programming Interface</p>
</def>
</def-item>
<def-item>
<term>BERT</term>
<def>
<p>Bidirectional Encoder Representations from Transformers</p>
</def>
</def-item>
<def-item>
<term>BiLSTM</term>
<def>
<p>Bidirectional Long Short-Term Memory</p>
</def>
</def-item>
<def-item>
<term>BoW</term>
<def>
<p>Bag of Words</p>
</def>
</def-item>
<def-item>
<term>CAF-I</term>
<def>
<p>Collaborative Agent Framework for Irony Detection</p>
</def>
</def-item>
<def-item>
<term>CBOW</term>
<def>
<p>Continuous Bag-of-Words</p>
</def>
</def-item>
<def-item>
<term>CIR</term>
<def>
<p>Classification Instability Rate</p>
</def>
</def-item>
<def-item>
<term>CMU-MOSEI</term>
<def>
<p>CMU Multimodal Opinion Sentiment and Emotion Intensity</p>
</def>
</def-item>
<def-item>
<term>CMU-MOSI</term>
<def>
<p>CMU Multimodal Opinion Sentiment Intensity</p>
</def>
</def-item>
<def-item>
<term>CNN</term>
<def>
<p>Convolutional Neural Network</p>
</def>
</def-item>
<def-item>
<term>CoT</term>
<def>
<p>Chain-of-Thought</p>
</def>
</def-item>
<def-item>
<term>DP</term>
<def>
<p>Differential Privacy</p>
</def>
</def-item>
<def-item>
<term>FL</term>
<def>
<p>Federated Learning</p>
</def>
</def-item>
<def-item>
<term>FLOPs</term>
<def>
<p>Floating-Point Operations</p>
</def>
</def-item>
<def-item>
<term>FTC</term>
<def>
<p>Federal Trade Commission</p>
</def>
</def-item>
<def-item>
<term>GloVe</term>
<def>
<p>Global Vectors for Word Representation</p>
</def>
</def-item>
<def-item>
<term>GQA</term>
<def>
<p>Grouped Query Attention</p>
</def>
</def-item>
<def-item>
<term>GRU</term>
<def>
<p>Gated Recurrent Unit</p>
</def>
</def-item>
<def-item>
<term>HAN</term>
<def>
<p>Hierarchical Attention Network</p>
</def>
</def-item>
<def-item>
<term>IMDb</term>
<def>
<p>Internet Movie Database</p>
</def>
</def-item>
<def-item>
<term>LACA</term>
<def>
<p>LLM-Augmented Cross-lingual ABSA</p>
</def>
</def-item>
<def-item>
<term>LFD-RT</term>
<def>
<p>Language-Family-Driven Resource Transfer</p>
</def>
</def-item>
<def-item>
<term>LIME</term>
<def>
<p>Local Interpretable Model-agnostic Explanations</p>
</def>
</def-item>
<def-item>
<term>LIWC</term>
<def>
<p>Linguistic Inquiry and Word Count</p>
</def>
</def-item>
<def-item>
<term>LLM</term>
<def>
<p>Large Language Model</p>
</def>
</def-item>
<def-item>
<term>LoRA</term>
<def>
<p>Low-Rank Adaptation</p>
</def>
</def-item>
<def-item>
<term>LSTM</term>
<def>
<p>Long Short-Term Memory</p>
</def>
</def-item>
<def-item>
<term>MAE</term>
<def>
<p>Mean Absolute Error</p>
</def>
</def-item>
<def-item>
<term>MELD</term>
<def>
<p>Multimodal EmotionLines Dataset</p>
</def>
</def-item>
<def-item>
<term>ML</term>
<def>
<p>Machine Learning</p>
</def>
</def-item>
<def-item>
<term>MLM</term>
<def>
<p>Masked Language Modeling</p>
</def>
</def-item>
<def-item>
<term>MLLM</term>
<def>
<p>Multimodal Large Language Model</p>
</def>
</def-item>
<def-item>
<term>MLP</term>
<def>
<p>Multi-Layer Perceptron</p>
</def>
</def-item>
<def-item>
<term>MMBERT</term>
<def>
<p>Multimodal BERT</p>
</def>
</def-item>
<def-item>
<term>MoE</term>
<def>
<p>Mixture of Experts</p>
</def>
</def-item>
<def-item>
<term>MSA</term>
<def>
<p>Multimodal Sentiment Analysis</p>
</def>
</def-item>
<def-item>
<term>MVP</term>
<def>
<p>Model Variability Problem</p>
</def>
</def-item>
<def-item>
<term>NLP</term>
<def>
<p>Natural Language Processing</p>
</def>
</def-item>
<def-item>
<term>NSP</term>
<def>
<p>Next Sentence Prediction</p>
</def>
</def-item>
<def-item>
<term>PLM</term>
<def>
<p>Pre-trained Language Model</p>
</def>
</def-item>
<def-item>
<term>QLoRA</term>
<def>
<p>Quantized LoRA</p>
</def>
</def-item>
<def-item>
<term>ReLU</term>
<def>
<p>Rectified Linear Unit</p>
</def>
</def-item>
<def-item>
<term>RNN</term>
<def>
<p>Recurrent Neural Network</p>
</def>
</def-item>
<def-item>
<term>RoBERTa</term>
<def>
<p>Robustly Optimized BERT Pretraining Approach</p>
</def>
</def-item>
<def-item>
<term>RoPE</term>
<def>
<p>Rotary Positional Embeddings</p>
</def>
</def-item>
<def-item>
<term>SA</term>
<def>
<p>Sentiment Analysis</p>
</def>
</def-item>
<def-item>
<term>SAoT</term>
<def>
<p>Sentiment Analysis of Thought</p>
</def>
</def-item>
<def-item>
<term>SGD</term>
<def>
<p>Stochastic Gradient Descent</p>
</def>
</def-item>
<def-item>
<term>SHAP</term>
<def>
<p>SHapley Additive exPlanations</p>
</def>
</def-item>
<def-item>
<term>SLM</term>
<def>
<p>Small Language Model</p>
</def>
</def-item>
<def-item>
<term>SST</term>
<def>
<p>Stanford Sentiment Treebank</p>
</def>
</def-item>
<def-item>
<term>SVM</term>
<def>
<p>Support Vector Machine</p>
</def>
</def-item>
<def-item>
<term>TF-IDF</term>
<def>
<p>Term Frequency&#x2013;Inverse Document Frequency</p>
</def>
</def-item>
<def-item>
<term>THOR</term>
<def>
<p>Three-Hop Reasoning</p>
</def>
</def-item>
<def-item>
<term>XAI</term>
<def>
<p>Explainable Artificial Intelligence</p>
</def>
</def-item>
<def-item>
<term>XLM-R</term>
<def>
<p>Cross-lingual Language Model&#x2013;RoBERTa</p>
</def>
</def-item>
</def-list>
</glossary>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Pang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>L</given-names></string-name>, <string-name><surname>Vaithyanathan</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Thumbs up? Sentiment classification using machine learning techniques</article-title>. In: <conf-name>Proceedings of the ACL-02 Conference on Empirical Methods in Natural Language Processing&#x2014;EMNLP; 2002 Jul 6&#x2013;7; Philadelphia, PA, USA</conf-name>. p. <fpage>79</fpage>&#x2013;<lpage>86</lpage>. doi:<pub-id pub-id-type="doi">10.3115/1118693.1118704</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>B</given-names></string-name></person-group>. <article-title>Deep learning for sentiment analysis: a survey</article-title>. <source>WIREs Data Min Knowl</source>. <year>2018</year>;<volume>8</volume>(<issue>4</issue>):<fpage>e1253</fpage>. doi:<pub-id pub-id-type="doi">10.1002/widm.1253</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Pang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Opinion mining and sentiment analysis</article-title>. <source>Found Trends in Inf Retr</source>. <year>2008</year>;<volume>2</volume>(<issue>1&#x2013;2</issue>):<fpage>1</fpage>&#x2013;<lpage>135</lpage>. doi:<pub-id pub-id-type="doi">10.1561/1500000011</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wankhade</surname> <given-names>M</given-names></string-name>, <string-name><surname>Rao</surname> <given-names>ACS</given-names></string-name>, <string-name><surname>Kulkarni</surname> <given-names>C</given-names></string-name></person-group>. <article-title>A survey on sentiment analysis methods, applications, and challenges</article-title>. <source>Artif Intell Rev</source>. <year>2022</year>;<volume>55</volume>(<issue>7</issue>):<fpage>5731</fpage>&#x2013;<lpage>80</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10462-022-10144-1</pub-id>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Birjali</surname> <given-names>M</given-names></string-name>, <string-name><surname>Kasri</surname> <given-names>M</given-names></string-name>, <string-name><surname>Beni-Hssane</surname> <given-names>A</given-names></string-name></person-group>. <article-title>A comprehensive survey on sentiment analysis: approaches, challenges and trends</article-title>. <source>Knowl Based Syst</source>. <year>2021</year>;<volume>226</volume>:<fpage>107134</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.knosys.2021.107134</pub-id>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Baccianella</surname> <given-names>S</given-names></string-name>, <string-name><surname>Esuli</surname> <given-names>A</given-names></string-name>, <string-name><surname>Sebastiani</surname> <given-names>F</given-names></string-name></person-group>. <article-title>SentiWordNet 3.0: an enhanced lexical resource for sentiment analysis and opinion mining</article-title>. In: <conf-name>Proceedings of the International Conference on Language Resources and Evaluation (LREC); 2010 May 17&#x2013;23; Valletta, Malta</conf-name>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Esuli</surname> <given-names>A</given-names></string-name>, <string-name><surname>Sebastiani</surname> <given-names>F</given-names></string-name></person-group>. <article-title>SentiWordNet: a publicly available lexical resource for opinion mining</article-title>. In: <conf-name>Proceedings of the International Conference on Language Resources and Evaluation (LREC); 2006 May 22&#x2013;28; Genoa, Italy</conf-name>. p. <fpage>417</fpage>&#x2013;<lpage>22</lpage>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hutto</surname> <given-names>C</given-names></string-name>, <string-name><surname>Gilbert</surname> <given-names>E</given-names></string-name></person-group>. <article-title>VADER: a parsimonious rule-based model for sentiment analysis of social media text</article-title>. <source>Proc Int AAAI Conf Web Soc Medium</source>. <year>2014</year>;<volume>8</volume>(<issue>1</issue>):<fpage>216</fpage>&#x2013;<lpage>25</lpage>. doi:<pub-id pub-id-type="doi">10.1609/icwsm.v8i1.14550</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Joachims</surname> <given-names>T</given-names></string-name></person-group>. <chapter-title>Text categorization with support vector machines: learning with many relevant features</chapter-title>. In: <source>Machine learning: ECML-98</source>. <publisher-loc>Berlin/Heidelberg, Germany</publisher-loc>: <publisher-name>Springer</publisher-name>; <year>1998</year>. p. <fpage>137</fpage>&#x2013;<lpage>42</lpage>. doi:<pub-id pub-id-type="doi">10.1007/BFb0026683</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>McCallum</surname> <given-names>A</given-names></string-name>, <string-name><surname>Nigam</surname> <given-names>K</given-names></string-name></person-group>. <article-title>A comparison of event models for naive Bayes text classification</article-title>. In: <conf-name>Proceedings of the AAAI-98 Workshop on Learning for Text Categorization; 1998 Jul 26&#x2013;27; Madison, WI, USA</conf-name>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Turney</surname> <given-names>PD</given-names></string-name></person-group>. <article-title>Thumbs up or thumbs down? Semantic orientation applied to unsupervised classification of reviews</article-title>. In: <conf-name>Proceedings of the 40th Annual Meeting on Association for Computational Linguistics&#x2014;ACL; 2002 Jul 7&#x2013;12; Philadelphia, PA, USA</conf-name>. p. <fpage>417</fpage>&#x2013;<lpage>24</lpage>. doi:<pub-id pub-id-type="doi">10.3115/1073083.1073153</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Kim</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Convolutional neural networks for sentence classification</article-title>. In: <conf-name>Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP); 2014 Oct 25&#x2013;29; Doha, Qatar</conf-name>. p. <fpage>1746</fpage>&#x2013;<lpage>51</lpage>. doi:<pub-id pub-id-type="doi">10.3115/v1/d14-1181</pub-id>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hochreiter</surname> <given-names>S</given-names></string-name>, <string-name><surname>Schmidhuber</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Long short-term memory</article-title>. <source>Neural Comput</source>. <year>1997</year>;<volume>9</volume>(<issue>8</issue>):<fpage>1735</fpage>&#x2013;<lpage>80</lpage>. doi:<pub-id pub-id-type="doi">10.1162/neco.1997.9.8.1735</pub-id>; <pub-id pub-id-type="pmid">9377276</pub-id></mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Bahdanau</surname> <given-names>D</given-names></string-name>, <string-name><surname>Cho</surname> <given-names>K</given-names></string-name>, <string-name><surname>Bengio</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Neural machine translation by jointly learning to align and translate</article-title>. In: <conf-name>Proceedings of the International Conference on Learning Representations (ICLR); 2015 May 7&#x2013;9; San Diego, CA, USA</conf-name>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Attention-based LSTM for aspect-level sentiment classification</article-title>. In: <conf-name>Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing; 2016 Nov 1&#x2013;5; Austin, TX, USA</conf-name>. p. <fpage>606</fpage>&#x2013;<lpage>15</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/d16-1058</pub-id>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Devlin</surname> <given-names>J</given-names></string-name>, <string-name><surname>Chang</surname> <given-names>MW</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>K</given-names></string-name>, <string-name><surname>Toutanova</surname> <given-names>K</given-names></string-name></person-group>. <article-title>BERT: pre-training of deep bidirectional transformers for language understanding</article-title>. In: <conf-name>Proceedings of the North American Chapter of the Association for Computational Linguistics (NAACL-HLT); 2019 Jun 2&#x2013;7; Minneapolis, MN, USA</conf-name>. p. <fpage>4171</fpage>&#x2013;<lpage>86</lpage>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ott</surname> <given-names>M</given-names></string-name>, <string-name><surname>Goyal</surname> <given-names>N</given-names></string-name>, <string-name><surname>Du</surname> <given-names>J</given-names></string-name>, <string-name><surname>Joshi</surname> <given-names>M</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>D</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>RoBERTa: a robustly optimized BERT pretraining approach</article-title>. <comment>arXiv:190711692. 2019</comment>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Dai</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Carbonell</surname> <given-names>J</given-names></string-name>, <string-name><surname>Salakhutdinov</surname> <given-names>R</given-names></string-name>, <string-name><surname>Le</surname> <given-names>QV</given-names></string-name></person-group>. <chapter-title>XLNet: generalized autoregressive pretraining for language understanding</chapter-title>. In: <source>Advances in neural information processing systems (NeurIPS)</source>. Vol. <volume>32</volume>. <publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc.</publisher-name>; <year>2019</year>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Sun</surname> <given-names>C</given-names></string-name>, <string-name><surname>Qiu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>X</given-names></string-name></person-group>. <chapter-title>How to fine-tune BERT for text classification?</chapter-title> In: <source>Chinese computational linguistics</source>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>; <year>2019</year>. p. <fpage>194</fpage>&#x2013;<lpage>206</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-030-32381-3_16</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Brown</surname> <given-names>T</given-names></string-name>, <string-name><surname>Mann</surname> <given-names>B</given-names></string-name>, <string-name><surname>Ryder</surname> <given-names>N</given-names></string-name>, <string-name><surname>Subbiah</surname> <given-names>M</given-names></string-name>, <string-name><surname>Kaplan</surname> <given-names>JD</given-names></string-name>, <string-name><surname>Dhariwal</surname> <given-names>P</given-names></string-name>, <etal>et al</etal></person-group>. <chapter-title>Language models are few-shot learners</chapter-title>. In: <source>Advances in neural information processing systems (NeurIPS)</source>. Vol. <volume>33</volume>. <publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc</publisher-name>.; <year>2020</year>. p. <fpage>1877</fpage>&#x2013;<lpage>901</lpage>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Grattafiori</surname> <given-names>A</given-names></string-name>, <string-name><surname>Dubey</surname> <given-names>A</given-names></string-name>, <string-name><surname>Jauhri</surname> <given-names>A</given-names></string-name>, <string-name><surname>Pandey</surname> <given-names>A</given-names></string-name>, <string-name><surname>Kadian</surname> <given-names>A</given-names></string-name>, <string-name><surname>Al-Dahle</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>The Llama 3 herd of models</article-title>. <comment>arXiv:240721783. 2024</comment>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Deng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>B</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Bing</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Sentiment analysis in the era of large language models: a reality check</article-title>. In: <conf-name>Proceedings of the Findings of the Association for Computational Linguistics: NAACL 2024; 2024 Jun 16&#x2013;21; Mexico City, Mexico</conf-name>. p. <fpage>3881</fpage>&#x2013;<lpage>906</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2024.findings-naacl.246</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Krugmann</surname> <given-names>JO</given-names></string-name>, <string-name><surname>Hartmann</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Sentiment analysis in the age of generative AI</article-title>. <source>Cust Needs Solut</source>. <year>2024</year>;<volume>11</volume>(<issue>1</issue>):<fpage>3</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s40547-024-00143-4</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Rathje</surname> <given-names>S</given-names></string-name>, <string-name><surname>Mirea</surname> <given-names>DM</given-names></string-name>, <string-name><surname>Sucholutsky</surname> <given-names>I</given-names></string-name>, <string-name><surname>Marjieh</surname> <given-names>R</given-names></string-name>, <string-name><surname>Robertson</surname> <given-names>CE</given-names></string-name>, <string-name><surname>Van Bavel</surname> <given-names>JJ</given-names></string-name></person-group>. <article-title>GPT is an effective tool for multilingual psychological text analysis</article-title>. <source>Proc Natl Acad Sci U S A</source>. <year>2024</year>;<volume>121</volume>(<issue>34</issue>):<fpage>e2308950121</fpage>. doi:<pub-id pub-id-type="doi">10.1073/pnas.2308950121</pub-id>; <pub-id pub-id-type="pmid">39133853</pub-id></mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Herrera-Poyatos</surname> <given-names>D</given-names></string-name>, <string-name><surname>Pel&#x00E1;ez-Gonz&#x00E1;lez</surname> <given-names>C</given-names></string-name>, <string-name><surname>Zuheros</surname> <given-names>C</given-names></string-name>, <string-name><surname>Herrera-Poyatos</surname> <given-names>A</given-names></string-name>, <string-name><surname>Tejedor</surname> <given-names>V</given-names></string-name>, <string-name><surname>Herrera</surname> <given-names>F</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>An overview of model uncertainty and variability in LLM-based sentiment analysis: challenges, mitigation strategies, and the role of explainability</article-title>. <source>Front Artif Intell</source>. <year>2025</year>;<volume>8</volume>:<fpage>1609097</fpage>. doi:<pub-id pub-id-type="doi">10.3389/frai.2025.1609097</pub-id>; <pub-id pub-id-type="pmid">40860717</pub-id></mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Joshi</surname> <given-names>A</given-names></string-name>, <string-name><surname>Bhattacharyya</surname> <given-names>P</given-names></string-name>, <string-name><surname>Carman</surname> <given-names>MJ</given-names></string-name></person-group>. <article-title>Automatic sarcasm detection: a survey</article-title>. <source>ACM Comput Surv</source>. <year>2018</year>;<volume>50</volume>(<issue>5</issue>):<fpage>1</fpage>&#x2013;<lpage>22</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3124420</pub-id>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ghosh</surname> <given-names>A</given-names></string-name>, <string-name><surname>Li</surname> <given-names>G</given-names></string-name>, <string-name><surname>Veale</surname> <given-names>T</given-names></string-name>, <string-name><surname>Rosso</surname> <given-names>P</given-names></string-name>, <string-name><surname>Shutova</surname> <given-names>E</given-names></string-name>, <string-name><surname>Barnden</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>SemEval-2015 task 11:sentiment analysis of figurative language in twitter</article-title>. In: <conf-name>Proceedings of the 9th International Workshop on Semantic Evaluation (SemEval 2015); 2015 Jun 4&#x2013;5; Denver, CO, USA</conf-name>. p. <fpage>470</fpage>&#x2013;<lpage>8</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/s15-2080</pub-id>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zou</surname> <given-names>C</given-names></string-name>, <string-name><surname>Lian</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Tiwari</surname> <given-names>P</given-names></string-name>, <string-name><surname>Qin</surname> <given-names>J</given-names></string-name></person-group>. <article-title>SarcasmBench: towards evaluating large language models on sarcasm understanding</article-title>. <source>IEEE Trans Affective Comput</source>. <year>2025</year>;<volume>16</volume>(<issue>4</issue>):<fpage>2560</fpage>&#x2013;<lpage>78</lpage>. doi:<pub-id pub-id-type="doi">10.1109/taffc.2025.3604806</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Blitzer</surname> <given-names>J</given-names></string-name>, <string-name><surname>Dredze</surname> <given-names>M</given-names></string-name>, <string-name><surname>Pereira</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Biographies, bollywood, boom-boxes and blenders: domain adaptation for sentiment classification</article-title>. In: <conf-name>Proceedings of the Annual Meeting of the Association for Computational Linguistics (ACL); 2007 Jun 24&#x2013;29; Prague, Czech Republic</conf-name>. p. <fpage>440</fpage>&#x2013;<lpage>7</lpage>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Glorot</surname> <given-names>X</given-names></string-name>, <string-name><surname>Bordes</surname> <given-names>A</given-names></string-name>, <string-name><surname>Bengio</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Domain adaptation for large-scale sentiment classification: a deep learning approach</article-title>. In: <conf-name>Proceedings of the International Conference on Machine Learning (ICML); 2011 Jun 28&#x2013;Jul 2; Bellevue, WA, USA</conf-name>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Banea</surname> <given-names>C</given-names></string-name>, <string-name><surname>Mihalcea</surname> <given-names>R</given-names></string-name>, <string-name><surname>Wiebe</surname> <given-names>J</given-names></string-name>, <string-name><surname>Hassan</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Multilingual subjectivity analysis using machine translation</article-title>. In: <conf-name>Proceedings of the Conference on Empirical Methods in Natural Language Processing&#x2014;EMNLP; 2008 Oct 25&#x2013;27; Honolulu, HI, USA</conf-name>. p. <fpage>127</fpage>&#x2013;<lpage>35</lpage>. doi:<pub-id pub-id-type="doi">10.3115/1613715.1613734</pub-id>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Conneau</surname> <given-names>A</given-names></string-name>, <string-name><surname>Khandelwal</surname> <given-names>K</given-names></string-name>, <string-name><surname>Goyal</surname> <given-names>N</given-names></string-name>, <string-name><surname>Chaudhary</surname> <given-names>V</given-names></string-name>, <string-name><surname>Wenzek</surname> <given-names>G</given-names></string-name>, <string-name><surname>Guzm&#x00E1;n</surname> <given-names>F</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Unsupervised cross-lingual representation learning at scale</article-title>. In: <conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics; 2020 Jul 5&#x2013;10; Online</conf-name>. p. <fpage>8440</fpage>&#x2013;<lpage>51</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.747</pub-id>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>&#x0160;m&#x00ED;d</surname> <given-names>J</given-names></string-name>, <string-name><surname>Kr&#x00E1;l</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Cross-lingual aspect-based sentiment analysis: a survey on tasks, approaches, and challenges</article-title>. <source>Inf Fusion</source>. <year>2025</year>;<volume>120</volume>(<issue>2010</issue>):<fpage>103073</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.inffus.2025.103073</pub-id>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Poria</surname> <given-names>S</given-names></string-name>, <string-name><surname>Cambria</surname> <given-names>E</given-names></string-name>, <string-name><surname>Bajpai</surname> <given-names>R</given-names></string-name>, <string-name><surname>Hussain</surname> <given-names>A</given-names></string-name></person-group>. <article-title>A review of affective computing: from unimodal analysis to multimodal fusion</article-title>. <source>Inf Fusion</source>. <year>2017</year>;<volume>37</volume>(<issue>95&#x2013;110</issue>):<fpage>98</fpage>&#x2013;<lpage>125</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.inffus.2017.02.003</pub-id>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zadeh</surname> <given-names>A</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>M</given-names></string-name>, <string-name><surname>Poria</surname> <given-names>S</given-names></string-name>, <string-name><surname>Cambria</surname> <given-names>E</given-names></string-name>, <string-name><surname>Morency</surname> <given-names>LP</given-names></string-name></person-group>. <article-title>Tensor fusion network for multimodal sentiment analysis</article-title>. In: <conf-name>Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing; 2017 Sep 7&#x2013;11; Copenhagen, Denmark</conf-name>. p. <fpage>1103</fpage>&#x2013;<lpage>14</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/d17-1115</pub-id>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Large language models meet text-centric multimodal sentiment analysis: a survey</article-title>. <source>Sci China Inf Sci</source>. <year>2025</year>;<volume>68</volume>(<issue>10</issue>):<fpage>200101</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s11432-024-4593-8</pub-id>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Narayan</surname> <given-names>A</given-names></string-name>, <string-name><surname>Madhu Kumar</surname> <given-names>SD</given-names></string-name>, <string-name><surname>Chacko</surname> <given-names>AM</given-names></string-name></person-group>. <article-title>Trust at risk: detecting misinformation in LLM-generated product reviews and its implications for consumer behavior and platform governance</article-title>. <source>Telematics Inform Rep</source>. <year>2026</year>;<volume>21</volume>(<issue>12</issue>):<fpage>100285</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.teler.2025.100285</pub-id>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Xylogiannopoulos</surname> <given-names>KF</given-names></string-name>, <string-name><surname>Xanthopoulos</surname> <given-names>P</given-names></string-name>, <string-name><surname>Karampelas</surname> <given-names>P</given-names></string-name>, <string-name><surname>Bakamitsos</surname> <given-names>GA</given-names></string-name></person-group>. <article-title>ChatGPT paraphrased product reviews can confuse consumers and undermine their trust in genuine reviews</article-title>. <source>Can you tell the difference? Inf Process Manag</source>. <year>2024</year>;<volume>61</volume>(<issue>6</issue>):<fpage>103842</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.ipm.2024.103842</pub-id>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Jain</surname> <given-names>PK</given-names></string-name>, <string-name><surname>Pamula</surname> <given-names>R</given-names></string-name>, <string-name><surname>Srivastava</surname> <given-names>G</given-names></string-name></person-group>. <article-title>A systematic literature review on machine learning applications for consumer sentiment analysis using online reviews</article-title>. <source>Comput Sci Rev</source>. <year>2021</year>;<volume>41</volume>(<issue>1</issue>):<fpage>100413</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.cosrev.2021.100413</pub-id>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Raghunathan</surname> <given-names>N</given-names></string-name>, <string-name><surname>Saravanakumar</surname> <given-names>K</given-names></string-name></person-group>. <article-title>Challenges and issues in sentiment analysis: a comprehensive survey</article-title>. <source>IEEE Access</source>. <year>2023</year>;<volume>11</volume>:<fpage>69626</fpage>&#x2013;<lpage>42</lpage>. doi:<pub-id pub-id-type="doi">10.1109/access.2023.3293041</pub-id>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Tetteh</surname> <given-names>M</given-names></string-name>, <string-name><surname>Thushara</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Sentiment analysis tools for movie review evaluation&#x2014;a survey</article-title>. In: <conf-name>Proceedings of the 2023 7th International Conference on Intelligent Computing and Control Systems (ICICCS); 2023 May 17&#x2013;19; Madurai, India</conf-name>. p. <fpage>816</fpage>&#x2013;<lpage>23</lpage>. doi:<pub-id pub-id-type="doi">10.1109/iciccs56967.2023.10142834</pub-id>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Islam</surname> <given-names>MS</given-names></string-name>, <string-name><surname>Kabir</surname> <given-names>MN</given-names></string-name>, <string-name><surname>Ghani</surname> <given-names>NA</given-names></string-name>, <string-name><surname>Zamli</surname> <given-names>KZ</given-names></string-name>, <string-name><surname>Zulkifli</surname> <given-names>NSA</given-names></string-name>, <string-name><surname>Rahman</surname> <given-names>MM</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Challenges and future in deep learning for sentiment analysis: a comprehensive review and a proposed novel hybrid approach</article-title>. <source>Artif Intell Rev</source>. <year>2024</year>;<volume>57</volume>(<issue>3</issue>):<fpage>62</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s10462-023-10651-9</pub-id>.</mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bordoloi</surname> <given-names>M</given-names></string-name>, <string-name><surname>Biswas</surname> <given-names>SK</given-names></string-name></person-group>. <article-title>Sentiment analysis: a survey on design framework, applications and future Scopes</article-title>. <source>Artif Intell Rev</source>. <year>2023</year>;<volume>56</volume>(<issue>11</issue>):<fpage>12505</fpage>&#x2013;<lpage>60</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10462-023-10442-2</pub-id>; <pub-id pub-id-type="pmid">37362892</pub-id></mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Sentiment analysis methods, applications, and challenges: a systematic literature review</article-title>. <source>J King Saud Univ Comput Inf Sci</source>. <year>2024</year>;<volume>36</volume>(<issue>4</issue>):<fpage>102048</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.jksuci.2024.102048</pub-id>.</mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kumar</surname> <given-names>M</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>L</given-names></string-name>, <string-name><surname>Chang</surname> <given-names>HT</given-names></string-name></person-group>. <article-title>Evolving techniques in sentiment analysis: a comprehensive review</article-title>. <source>PeerJ Comput Sci</source>. <year>2025</year>;<volume>11</volume>(<issue>3</issue>):<fpage>e2592</fpage>. doi:<pub-id pub-id-type="doi">10.7717/peerj-cs.2592</pub-id>; <pub-id pub-id-type="pmid">39957863</pub-id></mixed-citation></ref>
<ref id="ref-46"><label>[46]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Alahmadi</surname> <given-names>K</given-names></string-name>, <string-name><surname>Alharbi</surname> <given-names>S</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name></person-group>. <article-title>Generalizing sentiment analysis: a review of progress, challenges, and emerging directions</article-title>. <source>Soc Netw Anal Min</source>. <year>2025</year>;<volume>15</volume>(<issue>1</issue>):<fpage>45</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s13278-025-01461-8</pub-id>.</mixed-citation></ref>
<ref id="ref-47"><label>[47]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Suryawanshi</surname> <given-names>NS</given-names></string-name></person-group>. <article-title>Sentiment analysis with machine learning and deep learning: a survey of techniques and applications</article-title>. <source>Int J Sci Res Arch</source>. <year>2024</year>;<volume>12</volume>(<issue>2</issue>):<fpage>5</fpage>&#x2013;<lpage>15</lpage>. doi:<pub-id pub-id-type="doi">10.30574/ijsra.2024.12.2.1205</pub-id>.</mixed-citation></ref>
<ref id="ref-48"><label>[48]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bachate</surname> <given-names>M</given-names></string-name>, <string-name><surname>Suchitra</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Sentiment analysis and emotion recognition in social media: a comprehensive survey</article-title>. <source>Appl Soft Comput</source>. <year>2025</year>;<volume>174</volume>(<issue>3</issue>):<fpage>112958</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.asoc.2025.112958</pub-id>.</mixed-citation></ref>
<ref id="ref-49"><label>[49]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hankar</surname> <given-names>M</given-names></string-name>, <string-name><surname>Mzili</surname> <given-names>T</given-names></string-name>, <string-name><surname>Kasri</surname> <given-names>M</given-names></string-name>, <string-name><surname>Beni-Hssane</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Sentiment analysis survey: datasets, techniques, applications, tools, and challenges</article-title>. <source>Knowl Inf Syst</source>. <year>2025</year>;<volume>67</volume>(<issue>10</issue>):<fpage>8219</fpage>&#x2013;<lpage>65</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10115-025-02499-y</pub-id>.</mixed-citation></ref>
<ref id="ref-50"><label>[50]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ahmad Alomari</surname> <given-names>E</given-names></string-name></person-group>. <article-title>Unlocking the potential: a comprehensive systematic review of ChatGPT in natural language processing tasks</article-title>. <source>Comput Model Eng Sci</source>. <year>2024</year>;<volume>141</volume>(<issue>1</issue>):<fpage>43</fpage>&#x2013;<lpage>85</lpage>. doi:<pub-id pub-id-type="doi">10.32604/cmes.2024.052256</pub-id>.</mixed-citation></ref>
<ref id="ref-51"><label>[51]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Danyal</surname> <given-names>MM</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>SS</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>M</given-names></string-name>, <string-name><surname>Ullah</surname> <given-names>S</given-names></string-name>, <string-name><surname>Mehmood</surname> <given-names>F</given-names></string-name>, <string-name><surname>Ali</surname> <given-names>I</given-names></string-name></person-group>. <article-title>Proposing sentiment analysis model based on BERT and XLNet for movie reviews</article-title>. <source>Multimed Tools Appl</source>. <year>2024</year>;<volume>83</volume>(<issue>24</issue>):<fpage>64315</fpage>&#x2013;<lpage>39</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s11042-024-18156-5</pub-id>.</mixed-citation></ref>
<ref id="ref-52"><label>[52]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Shen</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Financial sentiment analysis on news and reports using large language models and FinBERT</article-title>. <comment>arXiv:241001987. 2024</comment>.</mixed-citation></ref>
<ref id="ref-53"><label>[53]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Boji&#x0107;</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zagovora</surname> <given-names>O</given-names></string-name>, <string-name><surname>Zelenkauskaite</surname> <given-names>A</given-names></string-name>, <string-name><surname>Vukovi&#x0107;</surname> <given-names>V</given-names></string-name>, <string-name><surname>&#x010C;abarkapa</surname> <given-names>M</given-names></string-name>, <string-name><surname>Veseljevi&#x0107; Jerkovi&#x0107;</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Comparing large language models and human annotators in latent content analysis of sentiment, political leaning, emotional intensity and sarcasm</article-title>. <source>Sci Rep</source>. <year>2025</year>;<volume>15</volume>(<issue>1</issue>):<fpage>11477</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-025-96508-3</pub-id>; <pub-id pub-id-type="pmid">40181141</pub-id></mixed-citation></ref>
<ref id="ref-54"><label>[54]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Fei</surname> <given-names>H</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Bing</surname> <given-names>L</given-names></string-name>, <string-name><surname>Li</surname> <given-names>F</given-names></string-name>, <string-name><surname>Chua</surname> <given-names>TS</given-names></string-name></person-group>. <article-title>Reasoning implicit sentiment with chain-of-thought prompting</article-title>. In: <conf-name>Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics; 2023 Jul 9&#x2013;14; Toronto, ON, Canada</conf-name>. p. <fpage>1171</fpage>&#x2013;<lpage>82</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2023.acl-short.101</pub-id>.</mixed-citation></ref>
<ref id="ref-55"><label>[55]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Duan</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Implicit sentiment analysis based on chain of thought prompting</article-title>. <comment>arXiv:240812157. 2024</comment>.</mixed-citation></ref>
<ref id="ref-56"><label>[56]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Zheng</surname> <given-names>K</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Li</surname> <given-names>L</given-names></string-name></person-group>. <chapter-title>Reassessing the role of chain-of-thought in sentiment analysis: Insights and limitations</chapter-title>. In: <source>Advanced intelligent computing technology and applications</source>. <publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer Nature</publisher-name>; <year>2025</year>. p. <fpage>89</fpage>&#x2013;<lpage>100</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-981-95-0020-8_8</pub-id>.</mixed-citation></ref>
<ref id="ref-57"><label>[57]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>He</surname> <given-names>Y</given-names></string-name>, <string-name><surname>He</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Gu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Gu</surname> <given-names>B</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Multi-chain of thought prompt learning for aspect-based sentiment analysis</article-title>. <source>Appl Sci</source>. <year>2025</year>;<volume>15</volume>(<issue>22</issue>):<fpage>12225</fpage>. doi:<pub-id pub-id-type="doi">10.3390/app152212225</pub-id>.</mixed-citation></ref>
<ref id="ref-58"><label>[58]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Warner</surname> <given-names>B</given-names></string-name>, <string-name><surname>Chaffin</surname> <given-names>A</given-names></string-name>, <string-name><surname>Clavi&#x00E9;</surname> <given-names>B</given-names></string-name>, <string-name><surname>Weller</surname> <given-names>O</given-names></string-name>, <string-name><surname>Hallstr&#x00F6;m</surname> <given-names>O</given-names></string-name>, <string-name><surname>Taghadouini</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Longer: a modern bidirectional encoder for fast, memory efficient, and long context finetuning and inference</article-title>. <comment>arXiv:241213663. 2024</comment>.</mixed-citation></ref>
<ref id="ref-59"><label>[59]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>da Silva</surname> <given-names>NB</given-names></string-name>, <string-name><surname>Harrison</surname> <given-names>J</given-names></string-name>, <string-name><surname>Minetto</surname> <given-names>R</given-names></string-name>, <string-name><surname>Delgado</surname> <given-names>MR</given-names></string-name>, <string-name><surname>Nassu</surname> <given-names>BT</given-names></string-name>, <string-name><surname>Silva</surname> <given-names>TH</given-names></string-name></person-group>. <article-title>Do multimodal LLMs see sentiment? The MLLMsent framework</article-title>. <comment>arXiv:250816873. 2025</comment>.</mixed-citation></ref>
<ref id="ref-60"><label>[60]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Li</surname> <given-names>T</given-names></string-name></person-group>. <article-title>A review of multimodal sentiment analysis in online public opinion monitoring</article-title>. <source>Informatics</source>. <year>2026</year>;<volume>13</volume>(<issue>1</issue>):<fpage>10</fpage>. doi:<pub-id pub-id-type="doi">10.3390/informatics13010010</pub-id>.</mixed-citation></ref>
<ref id="ref-61"><label>[61]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Miah</surname> <given-names>MSU</given-names></string-name>, <string-name><surname>Kabir</surname> <given-names>MM</given-names></string-name>, <string-name><surname>Bin Sarwar</surname> <given-names>T</given-names></string-name>, <string-name><surname>Safran</surname> <given-names>M</given-names></string-name>, <string-name><surname>Alfarhood</surname> <given-names>S</given-names></string-name>, <string-name><surname>Mridha</surname> <given-names>MF</given-names></string-name></person-group>. <article-title>A multimodal approach to cross-lingual sentiment analysis with ensemble of transformer and LLM</article-title>. <source>Sci Rep</source>. <year>2024</year>;<volume>14</volume>(<issue>1</issue>):<fpage>9603</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-024-60210-7</pub-id>; <pub-id pub-id-type="pmid">38671064</pub-id></mixed-citation></ref>
<ref id="ref-62"><label>[62]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Gardiner</surname> <given-names>S</given-names></string-name>, <string-name><surname>Rold&#x00E1;n</surname> <given-names>T</given-names></string-name>, <string-name><surname>Rossouw</surname> <given-names>D</given-names></string-name></person-group>. <article-title>The model arena for cross-lingual sentiment analysis: a comparative study in the era of large language models</article-title>. In: <conf-name>Proceedings of the 14th Workshop on Computational Approaches to Subjectivity, Sentiment, &#x0026; Social Media Analysis; 2024 Aug 15; Bangkok, Thailand</conf-name>. p. <fpage>141</fpage>&#x2013;<lpage>52</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2024.wassa-1.12</pub-id>.</mixed-citation></ref>
<ref id="ref-63"><label>[63]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Guan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>WJ</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>C</given-names></string-name>, <string-name><surname>Guan</surname> <given-names>Z</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Cross-lingual multimodal sentiment analysis for low-resource languages via language family disentanglement and rethinking transfer</article-title>. In: <conf-name>Proceedings of the Findings of the Association for Computational Linguistics: ACL 2025; 2025 Jul 27&#x2013;Aug 1; Vienna, Austria</conf-name>. p. <fpage>6513</fpage>&#x2013;<lpage>22</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2025.findings-acl.338</pub-id>.</mixed-citation></ref>
<ref id="ref-64"><label>[64]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Shang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Bridging resource gaps in cross-lingual sentiment analysis: adaptive self-alignment with data augmentation and transfer learning</article-title>. <source>PeerJ Comput Sci</source>. <year>2025</year>;<volume>11</volume>(<issue>1</issue>):<fpage>e2851</fpage>. doi:<pub-id pub-id-type="doi">10.7717/peerj-cs.2851</pub-id>; <pub-id pub-id-type="pmid">40567725</pub-id></mixed-citation></ref>
<ref id="ref-65"><label>[65]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>&#x0160;m&#x00ED;d</surname> <given-names>J</given-names></string-name>, <string-name><surname>Priban</surname> <given-names>P</given-names></string-name>, <string-name><surname>Kral</surname> <given-names>P</given-names></string-name></person-group>. <article-title>LACA: improving cross-lingual aspect-based sentiment analysis with LLM data augmentation</article-title>. In: <conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics; 2025 Jul 27&#x2013;Aug 1; Vienna, Austria</conf-name>. p. <fpage>839</fpage>&#x2013;<lpage>53</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.41</pub-id>.</mixed-citation></ref>
<ref id="ref-66"><label>[66]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Z</given-names></string-name></person-group>. <chapter-title>CAF-I: a collaborative multi-agent framework for enhanced irony detection with large language models</chapter-title>. In: <source>Neural information processing</source>. <publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer Nature</publisher-name>; <year>2025</year>. p. <fpage>153</fpage>&#x2013;<lpage>68</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-981-95-4367-0_11</pub-id>.</mixed-citation></ref>
<ref id="ref-67"><label>[67]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Oprea</surname> <given-names>SV</given-names></string-name>, <string-name><surname>B&#x00E2;ra</surname> <given-names>A</given-names></string-name></person-group>. <article-title>LLM-as-a-judge for sarcasm detection using supervised fine-tuning of transformers</article-title>. <source>J King Saud Univ Comput Inf Sci</source>. <year>2025</year>;<volume>37</volume>(<issue>10</issue>):<fpage>357</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s44443-025-00379-7</pub-id>.</mixed-citation></ref>
<ref id="ref-68"><label>[68]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Prabhu</surname> <given-names>O</given-names></string-name>, <string-name><surname>Navada</surname> <given-names>SG</given-names></string-name></person-group>. <article-title>ModernBERT-XAI: a synergistic approach to sentiment analysis with layer-wise learning and SHAP-LIME interpretability</article-title>. <source>Syst Sci Control Eng</source>. <year>2025</year>;<volume>13</volume>(<issue>1</issue>):<fpage>2600795</fpage>. doi:<pub-id pub-id-type="doi">10.1080/21642583.2025.2600795</pub-id>.</mixed-citation></ref>
<ref id="ref-69"><label>[69]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Perikos</surname> <given-names>I</given-names></string-name>, <string-name><surname>Diamantopoulos</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Explainable aspect-based sentiment analysis using transformer models</article-title>. <source>Big Data Cogn Comput</source>. <year>2024</year>;<volume>8</volume>(<issue>11</issue>):<fpage>141</fpage>. doi:<pub-id pub-id-type="doi">10.3390/bdcc8110141</pub-id>.</mixed-citation></ref>
<ref id="ref-70"><label>[70]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Taj</surname> <given-names>S</given-names></string-name>, <string-name><surname>Daudpota</surname> <given-names>SM</given-names></string-name>, <string-name><surname>Imran</surname> <given-names>AS</given-names></string-name>, <string-name><surname>Kastrati</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>Aspect-based sentiment analysis for software requirements elicitation using fine-tuned bidirectional encoder representations from transformers and explainable artificial intelligence</article-title>. <source>Eng Appl Artif Intell</source>. <year>2025</year>;<volume>151</volume>(<issue>9</issue>):<fpage>110632</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.engappai.2025.110632</pub-id>.</mixed-citation></ref>
<ref id="ref-71"><label>[71]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mienye</surname> <given-names>ID</given-names></string-name>, <string-name><surname>Swart</surname> <given-names>TG</given-names></string-name></person-group>. <article-title>Ensemble large language models: a survey</article-title>. <source>Information</source>. <year>2025</year>;<volume>16</volume>(<issue>8</issue>):<fpage>688</fpage>. doi:<pub-id pub-id-type="doi">10.3390/info16080688</pub-id>.</mixed-citation></ref>
<ref id="ref-72"><label>[72]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>B</given-names></string-name></person-group>. <source>Sentiment analysis and opinion mining</source>. <publisher-loc>San Rafael, CA, USA</publisher-loc>: <publisher-name>Morgan &#x0026; Claypool Publishers</publisher-name>; <year>2012</year>.</mixed-citation></ref>
<ref id="ref-73"><label>[73]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Bhatia</surname> <given-names>P</given-names></string-name>, <string-name><surname>Ji</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Eisenstein</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Better document-level sentiment analysis from RST discourse parsing</article-title>. In: <conf-name>Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing; 2015 Sep 17&#x2013;21; Lisbon, Portugal</conf-name>. p. <fpage>2212</fpage>&#x2013;<lpage>8</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/d15-1263</pub-id>.</mixed-citation></ref>
<ref id="ref-74"><label>[74]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wilson</surname> <given-names>T</given-names></string-name>, <string-name><surname>Wiebe</surname> <given-names>J</given-names></string-name>, <string-name><surname>Hoffmann</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Recognizing contextual polarity in phrase-level sentiment analysis</article-title>. In: <conf-name>Proceedings of the Conference on Human Language Technology and Empirical Methods in Natural Language Processing&#x2014;HLT; 2005 Oct 6&#x2013;8; Vancouver, BC, Canada</conf-name>. p. <fpage>347</fpage>&#x2013;<lpage>54</lpage>. doi:<pub-id pub-id-type="doi">10.3115/1220575.1220619</pub-id>.</mixed-citation></ref>
<ref id="ref-75"><label>[75]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>McDonald</surname> <given-names>R</given-names></string-name>, <string-name><surname>Hannan</surname> <given-names>K</given-names></string-name>, <string-name><surname>Neylon</surname> <given-names>T</given-names></string-name>, <string-name><surname>Wells</surname> <given-names>M</given-names></string-name>, <string-name><surname>Reynar</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Structured models for fine-to-coarse sentiment analysis</article-title>. In: <conf-name>Proceedings of the Annual Meeting of the Association for Computational Linguistics (ACL); 2007 Jun 24&#x2013;29; Prague, Czech Republic</conf-name>. p. <fpage>432</fpage>&#x2013;<lpage>9</lpage>.</mixed-citation></ref>
<ref id="ref-76"><label>[76]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Thet</surname> <given-names>TT</given-names></string-name>, <string-name><surname>Na</surname> <given-names>JC</given-names></string-name>, <string-name><surname>Khoo</surname> <given-names>CSG</given-names></string-name></person-group>. <article-title>Aspect-based sentiment analysis of movie reviews on discussion boards</article-title>. <source>J Inf Sci</source>. <year>2010</year>;<volume>36</volume>(<issue>6</issue>):<fpage>823</fpage>&#x2013;<lpage>48</lpage>. doi:<pub-id pub-id-type="doi">10.1177/0165551510388123</pub-id>.</mixed-citation></ref>
<ref id="ref-77"><label>[77]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>B</given-names></string-name></person-group>. <article-title>Mining and summarizing customer reviews</article-title>. In: <conf-name>Proceedings of the Tenth ACM SIGKDD International Conference on Knowledge Discovery and Data Mining; 2004 Aug 22&#x2013;25; Seattle, WA, USA</conf-name>. p. <fpage>168</fpage>&#x2013;<lpage>77</lpage>. doi:<pub-id pub-id-type="doi">10.1145/1014052.1014073</pub-id>.</mixed-citation></ref>
<ref id="ref-78"><label>[78]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Horsa</surname> <given-names>OG</given-names></string-name>, <string-name><surname>Tune</surname> <given-names>KK</given-names></string-name></person-group>. <article-title>Aspect-based sentiment analysis for Afaan Oromoo movie reviews using machine learning techniques</article-title>. <source>Appl Comput Intell Soft Comput</source>. <year>2023</year>;<volume>2023</volume>(<issue>1</issue>):<fpage>3462691</fpage>. doi:<pub-id pub-id-type="doi">10.1155/2023/3462691</pub-id>.</mixed-citation></ref>
<ref id="ref-79"><label>[79]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Musa</surname> <given-names>A</given-names></string-name>, <string-name><surname>Adam</surname> <given-names>FM</given-names></string-name>, <string-name><surname>Ibrahim</surname> <given-names>U</given-names></string-name>, <string-name><surname>Zandam</surname> <given-names>AY</given-names></string-name></person-group>. <article-title>HauBERT: a transformer model for aspect-based sentiment analysis of Hausa-language movie reviews</article-title>. <source>Eng Proc</source>. <year>2025</year>;<volume>87</volume>(<issue>1</issue>):<fpage>43</fpage>. doi:<pub-id pub-id-type="doi">10.3390/engproc2025087043</pub-id>.</mixed-citation></ref>
<ref id="ref-80"><label>[80]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Scaria</surname> <given-names>K</given-names></string-name>, <string-name><surname>Gupta</surname> <given-names>H</given-names></string-name>, <string-name><surname>Goyal</surname> <given-names>S</given-names></string-name>, <string-name><surname>Sawant</surname> <given-names>S</given-names></string-name>, <string-name><surname>Mishra</surname> <given-names>S</given-names></string-name>, <string-name><surname>Baral</surname> <given-names>C</given-names></string-name></person-group>. <article-title>InstructABSA: instruction learning for aspect based sentiment analysis</article-title>. In: <conf-name>Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies; 2024 Jun 16&#x2013;21; Mexico City, Mexico</conf-name>. p. <fpage>720</fpage>&#x2013;<lpage>36</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2024.naacl-short.63</pub-id>.</mixed-citation></ref>
<ref id="ref-81"><label>[81]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Demszky</surname> <given-names>D</given-names></string-name>, <string-name><surname>Movshovitz-Attias</surname> <given-names>D</given-names></string-name>, <string-name><surname>Ko</surname> <given-names>J</given-names></string-name>, <string-name><surname>Cowen</surname> <given-names>A</given-names></string-name>, <string-name><surname>Nemade</surname> <given-names>G</given-names></string-name>, <string-name><surname>Ravi</surname> <given-names>S</given-names></string-name></person-group>. <article-title>GoEmotions: a dataset of fine-grained emotions</article-title>. In: <conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics; 2020 Jul 5&#x2013;10; Online</conf-name>. p. <fpage>4040</fpage>&#x2013;<lpage>54</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.372</pub-id>.</mixed-citation></ref>
<ref id="ref-82"><label>[82]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Barbieri</surname> <given-names>F</given-names></string-name>, <string-name><surname>Camacho-Collados</surname> <given-names>J</given-names></string-name>, <string-name><surname>Espinosa Anke</surname> <given-names>L</given-names></string-name>, <string-name><surname>Neves</surname> <given-names>L</given-names></string-name></person-group>. <article-title>TweetEval: unified benchmark and comparative evaluation for tweet classification</article-title>. In: <conf-name>Proceedings of the Findings of the Association for Computational Linguistics: EMNLP 2020; 2020 Nov 16&#x2013;20; Online</conf-name>. p. <fpage>1644</fpage>&#x2013;<lpage>50</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2020.findings-emnlp.148</pub-id>.</mixed-citation></ref>
<ref id="ref-83"><label>[83]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Bagher Zadeh</surname> <given-names>A</given-names></string-name>, <string-name><surname>Liang</surname> <given-names>PP</given-names></string-name>, <string-name><surname>Poria</surname> <given-names>S</given-names></string-name>, <string-name><surname>Cambria</surname> <given-names>E</given-names></string-name>, <string-name><surname>Morency</surname> <given-names>LP</given-names></string-name></person-group>. <article-title>Multimodal language analysis in the wild: CMU-MOSEI dataset and interpretable dynamic fusion graph</article-title>. In: <conf-name>Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics; 2018 Jul 15&#x2013;20; Melbourne, Australia</conf-name>, p. <fpage>2236</fpage>&#x2013;<lpage>46</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/p18-1208</pub-id>.</mixed-citation></ref>
<ref id="ref-84"><label>[84]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Socher</surname> <given-names>R</given-names></string-name>, <string-name><surname>Perelygin</surname> <given-names>A</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Chuang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Manning</surname> <given-names>CD</given-names></string-name>, <string-name><surname>Ng</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Recursive deep models for semantic compositionality over a sentiment treebank</article-title>. In: <conf-name>Proceedings of the 2013 Conference on Empirical Methods in Natural Language Processing; 2013 Oct 18&#x2013;21; Seattle, WA, USA</conf-name>. p. <fpage>1631</fpage>&#x2013;<lpage>42</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/d13-1170</pub-id>.</mixed-citation></ref>
<ref id="ref-85"><label>[85]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Maas</surname> <given-names>A</given-names></string-name>, <string-name><surname>Daly</surname> <given-names>RE</given-names></string-name>, <string-name><surname>Pham</surname> <given-names>PT</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Ng</surname> <given-names>AY</given-names></string-name>, <string-name><surname>Potts</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Learning word vectors for sentiment analysis</article-title>. In: <conf-name>Proceedings of ACL-HLT; 2011 Jun 19&#x2013;24; Portland, OR, USA</conf-name>. p. <fpage>142</fpage>&#x2013;<lpage>50</lpage>.</mixed-citation></ref>
<ref id="ref-86"><label>[86]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kralj Novak</surname> <given-names>P</given-names></string-name>, <string-name><surname>Smailovi&#x0107;</surname> <given-names>J</given-names></string-name>, <string-name><surname>Sluban</surname> <given-names>B</given-names></string-name>, <string-name><surname>Mozeti&#x010D;</surname> <given-names>I</given-names></string-name></person-group>. <article-title>Sentiment of emojis</article-title>. <source>PLoS One</source>. <year>2015</year>;<volume>10</volume>(<issue>12</issue>):<fpage>e0144296</fpage>. doi:<pub-id pub-id-type="doi">10.1371/journal.pone.0144296</pub-id>; <pub-id pub-id-type="pmid">26641093</pub-id></mixed-citation></ref>
<ref id="ref-87"><label>[87]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Bhatia</surname> <given-names>G</given-names></string-name>, <string-name><surname>Nagoudi</surname> <given-names>EMB</given-names></string-name>, <string-name><surname>Cavusoglu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Abdul-Mageed</surname> <given-names>M</given-names></string-name></person-group>. <article-title>FinTral: a family of GPT-4 level multimodal financial large language models</article-title>. In: <conf-name>Proceedings of the Findings of the Association for Computational Linguistics ACL 2024; 2024 Aug 11&#x2013;16; Bangkok, Thailand</conf-name>. p. <fpage>13064</fpage>&#x2013;<lpage>87</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2024.findings-acl.774</pub-id>.</mixed-citation></ref>
<ref id="ref-88"><label>[88]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ranasinghe</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zampieri</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Multilingual offensive language identification with cross-lingual embeddings</article-title>. In: <conf-name>Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP); 2020 Nov 16&#x2013;20; Online</conf-name>. p. <fpage>5838</fpage>&#x2013;<lpage>44</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2020.emnlp-main.470</pub-id>.</mixed-citation></ref>
<ref id="ref-89"><label>[89]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Pang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>L</given-names></string-name></person-group>. <article-title>A sentimental education: sentiment analysis using subjectivity summarization based on minimum cuts</article-title>. In: <conf-name>Proceedings of the 42nd Annual Meeting on Association for Computational Linguistics&#x2014;ACL; 2004 Jul 21&#x2013;26; Barcelona, Spain</conf-name>. p. <fpage>271</fpage>&#x2013;<lpage>8</lpage>. doi:<pub-id pub-id-type="doi">10.3115/1218955.1218990</pub-id>.</mixed-citation></ref>
<ref id="ref-90"><label>[90]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zadeh</surname> <given-names>A</given-names></string-name>, <string-name><surname>Liang</surname> <given-names>PP</given-names></string-name>, <string-name><surname>Poria</surname> <given-names>S</given-names></string-name>, <string-name><surname>Vij</surname> <given-names>P</given-names></string-name>, <string-name><surname>Cambria</surname> <given-names>E</given-names></string-name>, <string-name><surname>Morency</surname> <given-names>LP</given-names></string-name></person-group>. <article-title>Multi-attention recurrent network for human communication comprehension</article-title>. <source>Proc AAAI Conf Artif Intell</source>. <year>2018</year>;<volume>32</volume>(<issue>1</issue>):<fpage>5642</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1609/aaai.v32i1.12024</pub-id>.</mixed-citation></ref>
<ref id="ref-91"><label>[91]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lai</surname> <given-names>S</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>Multimodal sentiment analysis: a survey</article-title>. <source>Displays</source>. <year>2023</year>;<volume>80</volume>(<issue>2</issue>):<fpage>102563</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.displa.2023.102563</pub-id>.</mixed-citation></ref>
<ref id="ref-92"><label>[92]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Poria</surname> <given-names>S</given-names></string-name>, <string-name><surname>Majumder</surname> <given-names>N</given-names></string-name>, <string-name><surname>Mihalcea</surname> <given-names>R</given-names></string-name>, <string-name><surname>Hovy</surname> <given-names>E</given-names></string-name></person-group>. <article-title>Emotion recognition in conversation: research challenges, datasets, and recent advances</article-title>. <source>IEEE Access</source>. <year>2019</year>;<volume>7</volume>:<fpage>100943</fpage>&#x2013;<lpage>53</lpage>. doi:<pub-id pub-id-type="doi">10.1109/access.2019.2929050</pub-id>.</mixed-citation></ref>
<ref id="ref-93"><label>[93]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Pennebaker</surname> <given-names>JW</given-names></string-name>, <string-name><surname>Francis</surname> <given-names>ME</given-names></string-name>, <string-name><surname>Booth</surname> <given-names>RJ</given-names></string-name></person-group>. <source>Linguistic inquiry and word count: LIWC 2001</source>. <publisher-loc>Mahwah, NJ, USA</publisher-loc>: <publisher-name>Lawrence Erlbaum Associates</publisher-name>; <year>2001</year>.</mixed-citation></ref>
<ref id="ref-94"><label>[94]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tausczik</surname> <given-names>YR</given-names></string-name>, <string-name><surname>Pennebaker</surname> <given-names>JW</given-names></string-name></person-group>. <article-title>The psychological meaning of words: LIWC and computerized text analysis methods</article-title>. <source>J Lang Soc Psychol</source>. <year>2010</year>;<volume>29</volume>(<issue>1</issue>):<fpage>24</fpage>&#x2013;<lpage>54</lpage>. doi:<pub-id pub-id-type="doi">10.1177/0261927x09351676</pub-id>.</mixed-citation></ref>
<ref id="ref-95"><label>[95]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kennedy</surname> <given-names>A</given-names></string-name>, <string-name><surname>Inkpen</surname> <given-names>D</given-names></string-name></person-group>. <article-title>Sentiment classification of movie reviews using contextual valence shifters</article-title>. <source>Comput Intell</source>. <year>2006</year>;<volume>22</volume>(<issue>2</issue>):<fpage>110</fpage>&#x2013;<lpage>25</lpage>. doi:<pub-id pub-id-type="doi">10.1111/j.1467-8640.2006.00277.x</pub-id>.</mixed-citation></ref>
<ref id="ref-96"><label>[96]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Whitelaw</surname> <given-names>C</given-names></string-name>, <string-name><surname>Garg</surname> <given-names>N</given-names></string-name>, <string-name><surname>Argamon</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Using appraisal groups for sentiment analysis</article-title>. In: <conf-name>Proceedings of the 14th ACM International Conference on Information and Knowledge Management; 2005 Oct 31&#x2013;Nov 5; Bremen, Germany</conf-name>. p. <fpage>625</fpage>&#x2013;<lpage>31</lpage>. doi:<pub-id pub-id-type="doi">10.1145/1099554.1099714</pub-id>.</mixed-citation></ref>
<ref id="ref-97"><label>[97]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Manning</surname> <given-names>CD</given-names></string-name>, <string-name><surname>Raghavan</surname> <given-names>P</given-names></string-name>, <string-name><surname>Schtze</surname> <given-names>H</given-names></string-name></person-group>. <source>Introduction to information retrieval</source>. <publisher-loc>Cambridge, UK</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>; <year>2008</year>.</mixed-citation></ref>
<ref id="ref-98"><label>[98]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yadav</surname> <given-names>A</given-names></string-name>, <string-name><surname>Vishwakarma</surname> <given-names>DK</given-names></string-name></person-group>. <article-title>Sentiment analysis using deep learning architectures: a review</article-title>. <source>Artif Intell Rev</source>. <year>2020</year>;<volume>53</volume>(<issue>6</issue>):<fpage>4335</fpage>&#x2013;<lpage>85</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10462-019-09794-5</pub-id>.</mixed-citation></ref>
<ref id="ref-99"><label>[99]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Goldberg</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>A primer on neural network models for natural language processing</article-title>. <source>JAIR</source>. <year>2016</year>;<volume>57</volume>:<fpage>345</fpage>&#x2013;<lpage>420</lpage>. doi:<pub-id pub-id-type="doi">10.1613/jair.4992</pub-id>.</mixed-citation></ref>
<ref id="ref-100"><label>[100]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Mikolov</surname> <given-names>T</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>K</given-names></string-name>, <string-name><surname>Corrado</surname> <given-names>G</given-names></string-name>, <string-name><surname>Dean</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Efficient estimation of word representations in vector space</article-title>. <comment>arXiv:13013781. 2013</comment>.</mixed-citation></ref>
<ref id="ref-101"><label>[101]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Mikolov</surname> <given-names>T</given-names></string-name>, <string-name><surname>Sutskever</surname> <given-names>I</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>K</given-names></string-name>, <string-name><surname>Corrado</surname> <given-names>GS</given-names></string-name>, <string-name><surname>Dean</surname> <given-names>J</given-names></string-name></person-group>. <chapter-title>Distributed representations of words and phrases and their compositionality</chapter-title>. In: <source>Advances in neural information processing systems (NeurIPS)</source>. Vol. <volume>26</volume>. <publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc</publisher-name>.; <year>2013</year>.</mixed-citation></ref>
<ref id="ref-102"><label>[102]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Pennington</surname> <given-names>J</given-names></string-name>, <string-name><surname>Socher</surname> <given-names>R</given-names></string-name>, <string-name><surname>Manning</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Global vectors for word representation</article-title>. In: <conf-name>Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP); 2014 Oct 25&#x2013;29; Doha, Qatar</conf-name>. p. <fpage>1532</fpage>&#x2013;<lpage>43</lpage>. doi:<pub-id pub-id-type="doi">10.3115/v1/d14-1162</pub-id>.</mixed-citation></ref>
<ref id="ref-103"><label>[103]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bengio</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Simard</surname> <given-names>P</given-names></string-name>, <string-name><surname>Frasconi</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Learning long-term dependencies with gradient descent is difficult</article-title>. <source>IEEE Trans Neural Netw</source>. <year>1994</year>;<volume>5</volume>(<issue>2</issue>):<fpage>157</fpage>&#x2013;<lpage>66</lpage>. doi:<pub-id pub-id-type="doi">10.1109/72.279181</pub-id>; <pub-id pub-id-type="pmid">18267787</pub-id></mixed-citation></ref>
<ref id="ref-104"><label>[104]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Tai</surname> <given-names>KS</given-names></string-name>, <string-name><surname>Socher</surname> <given-names>R</given-names></string-name>, <string-name><surname>Manning</surname> <given-names>CD</given-names></string-name></person-group>. <article-title>Improved semantic representations from tree-structured long short-term memory networks</article-title>. In: <conf-name>Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing; 2015 Jul 26&#x2013;31; Beijing, China</conf-name>. p. <fpage>1556</fpage>&#x2013;<lpage>66</lpage>. doi:<pub-id pub-id-type="doi">10.3115/v1/p15-1150</pub-id>.</mixed-citation></ref>
<ref id="ref-105"><label>[105]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Nkhata</surname> <given-names>G</given-names></string-name>, <string-name><surname>Gauch</surname> <given-names>S</given-names></string-name>, <string-name><surname>Anjum</surname> <given-names>U</given-names></string-name>, <string-name><surname>Zhan</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Fine-tuning BERT with bidirectional LSTM for fine-grained movie reviews sentiment analysis</article-title>. <comment>arXiv:250220682. 2025</comment>.</mixed-citation></ref>
<ref id="ref-106"><label>[106]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><collab>Towards Data Science</collab></person-group>. <article-title>Transformer architecture illustration</article-title>. <year>2022 [cited 2026 Mar 17]</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://towardsdatascience.com/wp-content/uploads/2022/12/1B0q2ZLsUUw31eEImeVf3PQ.png">https://towardsdatascience.com/wp-content/uploads/2022/12/1B0q2ZLsUUw31eEImeVf3PQ.png</ext-link>.</mixed-citation></ref>
<ref id="ref-107"><label>[107]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Dyer</surname> <given-names>C</given-names></string-name>, <string-name><surname>He</surname> <given-names>X</given-names></string-name>, <string-name><surname>Smola</surname> <given-names>A</given-names></string-name>, <string-name><surname>Hovy</surname> <given-names>E</given-names></string-name></person-group>. <article-title>Hierarchical attention networks for document classification</article-title>. In: <conf-name>Proceedings of the 2016 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies; 2016 Jun 12&#x2013;17; San Diego, CA, USA</conf-name>. p. <fpage>1480</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/n16-1174</pub-id>.</mixed-citation></ref>
<ref id="ref-108"><label>[108]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Vaswani</surname> <given-names>A</given-names></string-name>, <string-name><surname>Shazeer</surname> <given-names>N</given-names></string-name>, <string-name><surname>Parmar</surname> <given-names>N</given-names></string-name>, <string-name><surname>Uszkoreit</surname> <given-names>J</given-names></string-name>, <string-name><surname>Jones</surname> <given-names>L</given-names></string-name>, <string-name><surname>Gomez</surname> <given-names>AN</given-names></string-name>, <etal>et al</etal></person-group>. <chapter-title>Attention is all you need</chapter-title>. In: <source>Advances in neural information processing systems (NeurIPS)</source>. Vol. <volume>30</volume>. <publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc.</publisher-name>; <year>2017</year>. p. <fpage>5998</fpage>&#x2013;<lpage>6008</lpage>.</mixed-citation></ref>
<ref id="ref-109"><label>[109]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><collab>The AI Summer</collab></person-group>. <article-title>Query, key, value attention mechanism diagram. 2022 [cited 2026 Mar 17]</article-title>. Available from: <ext-link ext-link-type="uri" xlink:href="https://theaisummer.com/static/56773616d30b9dcb31aa792f2d701276/3096d/key-query-value.png">https://theaisummer.com/static/56773616d30b9dcb31aa792f2d701276/3096d/key-query-value.png</ext-link>.</mixed-citation></ref>
<ref id="ref-110"><label>[110]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Kumar</surname> <given-names>S</given-names></string-name></person-group>. <article-title>BERT architecture diagram. 2023 [cited 2026 Mar 17]</article-title>. Available from: <ext-link ext-link-type="uri" xlink:href="https://sushant-kumar.com/blog/">https://sushant-kumar.com/blog/</ext-link>.</mixed-citation></ref>
<ref id="ref-111"><label>[111]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Sanh</surname> <given-names>V</given-names></string-name>, <string-name><surname>Debut</surname> <given-names>L</given-names></string-name>, <string-name><surname>Chaumond</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wolf</surname> <given-names>T</given-names></string-name></person-group>. <article-title>DistilBERT, a distilled version of BERT: Smaller, faster, cheaper and lighter</article-title>. <comment>arXiv:191001108. 2019</comment>.</mixed-citation></ref>
<ref id="ref-112"><label>[112]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bello</surname> <given-names>A</given-names></string-name>, <string-name><surname>Ng</surname> <given-names>SC</given-names></string-name>, <string-name><surname>Leung</surname> <given-names>MF</given-names></string-name></person-group>. <article-title>A BERT framework to sentiment analysis of tweets</article-title>. <source>Sensors</source>. <year>2023</year>;<volume>23</volume>(<issue>1</issue>):<fpage>506</fpage>. doi:<pub-id pub-id-type="doi">10.3390/s23010506</pub-id>; <pub-id pub-id-type="pmid">36617101</pub-id></mixed-citation></ref>
<ref id="ref-113"><label>[113]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Batra</surname> <given-names>H</given-names></string-name>, <string-name><surname>Punn</surname> <given-names>NS</given-names></string-name>, <string-name><surname>Sonbhadra</surname> <given-names>SK</given-names></string-name>, <string-name><surname>Agarwal</surname> <given-names>S</given-names></string-name></person-group>. <chapter-title>BERT-based sentiment analysis: a software engineering perspective</chapter-title>. In: <source>Database and expert systems applications</source>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name>; <year>2021</year>. p. <fpage>138</fpage>&#x2013;<lpage>48</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-030-86472-9_13</pub-id>.</mixed-citation></ref>
<ref id="ref-114"><label>[114]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Penha</surname> <given-names>G</given-names></string-name>, <string-name><surname>Hauff</surname> <given-names>C</given-names></string-name></person-group>. <article-title>What does BERT know about books, movies and music? Probing BERT for Conversational Recommendation</article-title>. In: <conf-name>Proceedings of the Fourteenth ACM Conference on Recommender Systems; 2020 Sep 22&#x2013;26; Virtual</conf-name>. p. <fpage>388</fpage>&#x2013;<lpage>97</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3383313.3412249</pub-id>.</mixed-citation></ref>
<ref id="ref-115"><label>[115]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Gao</surname> <given-names>T</given-names></string-name>, <string-name><surname>Fisch</surname> <given-names>A</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>D</given-names></string-name></person-group>. <article-title>Making pre-trained language models better few-shot learners</article-title>. In: <conf-name>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing; 2021 Aug 1&#x2013;6; Online</conf-name>. p. <fpage>3816</fpage>&#x2013;<lpage>30</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2021.acl-long.295</pub-id>.</mixed-citation></ref>
<ref id="ref-116"><label>[116]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Wei</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Schuurmans</surname> <given-names>D</given-names></string-name>, <string-name><surname>Bosma</surname> <given-names>M</given-names></string-name>, <string-name><surname>Xia</surname> <given-names>F</given-names></string-name>, <string-name><surname>Chi</surname> <given-names>EH</given-names></string-name>, <etal>et al</etal></person-group>. <chapter-title>Chain-of-thought prompting elicits reasoning in large language models</chapter-title>. In: <source>Advances in neural information processing systems (NeurIPS)</source>. Vol. <volume>35</volume>. <publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc.</publisher-name>; <year>2022</year>. p. <fpage>24824</fpage>&#x2013;<lpage>37</lpage>.</mixed-citation></ref>
<ref id="ref-117"><label>[117]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Yao</surname> <given-names>C</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Graph-enhanced implicit aspect-level sentiment analysis based on multi-prompt fusion</article-title>. <source>Sci Rep</source>. <year>2025</year>;<volume>15</volume>(<issue>1</issue>):<fpage>17460</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-025-02609-4</pub-id>; <pub-id pub-id-type="pmid">40394193</pub-id></mixed-citation></ref>
<ref id="ref-118"><label>[118]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Stilwell</surname> <given-names>S</given-names></string-name>, <string-name><surname>Inkpen</surname> <given-names>D</given-names></string-name></person-group>. <article-title>Explainable prompt-based approaches for sentiment analysis of movie reviews</article-title>. <source>Proc Can Conf Artif Intell</source>. <year>2024</year>. doi:<pub-id pub-id-type="doi">10.21428/594757db.faf9e091</pub-id>.</mixed-citation></ref>
<ref id="ref-119"><label>[119]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Gu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Han</surname> <given-names>X</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>M</given-names></string-name></person-group>. <article-title>PPT: pre-trained prompt tuning for few-shot learning</article-title>. In: <conf-name>Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics; 2022 May 22&#x2013;27; Dublin, Ireland</conf-name>. p. <fpage>8410</fpage>&#x2013;<lpage>23</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2022.acl-long.576</pub-id>.</mixed-citation></ref>
<ref id="ref-120"><label>[120]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cai</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>F</given-names></string-name>, <string-name><surname>Rao</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Multimodal sentiment analysis based on multi-layer feature fusion and multi-task learning</article-title>. <source>Sci Rep</source>. <year>2025</year>;<volume>15</volume>(<issue>1</issue>):<fpage>2126</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-025-85859-6</pub-id>; <pub-id pub-id-type="pmid">39821109</pub-id></mixed-citation></ref>
<ref id="ref-121"><label>[121]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Ren</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Multimodal sentiment analysis based on BERT and ResNet</article-title>. <comment>arXiv:241203625. 2024</comment>.</mixed-citation></ref>
<ref id="ref-122"><label>[122]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Khare</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Bagal</surname> <given-names>V</given-names></string-name>, <string-name><surname>Mathew</surname> <given-names>M</given-names></string-name>, <string-name><surname>Devi</surname> <given-names>A</given-names></string-name>, <string-name><surname>Priyakumar</surname> <given-names>UD</given-names></string-name>, <string-name><surname>Jawahar</surname> <given-names>CV</given-names></string-name></person-group>. <article-title>MMBERT: multimodal BERT pretraining for improved medical VQA</article-title>. In: <conf-name>Proceedings of the 2021 IEEE 18th International Symposium on Biomedical Imaging (ISBI); 2021 Apr 13&#x2013;16; Nice, France</conf-name>. p. <fpage>1033</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1109/isbi48211.2021.9434063</pub-id>.</mixed-citation></ref>
<ref id="ref-123"><label>[123]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Schwartz</surname> <given-names>R</given-names></string-name>, <string-name><surname>Dodge</surname> <given-names>J</given-names></string-name>, <string-name><surname>Smith</surname> <given-names>NA</given-names></string-name>, <string-name><surname>Etzioni</surname> <given-names>O</given-names></string-name></person-group>. <article-title>Green AI</article-title>. <source>Commun ACM</source>. <year>2020</year>;<volume>63</volume>(<issue>12</issue>):<fpage>54</fpage>&#x2013;<lpage>63</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3381831</pub-id>.</mixed-citation></ref>
<ref id="ref-124"><label>[124]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Kreuz</surname> <given-names>RJ</given-names></string-name></person-group>. <chapter-title>The use of verbal irony: cues and constraints</chapter-title>. In: <source>Metaphor</source>. <publisher-loc>London, UK</publisher-loc>: <publisher-name>Psychology Press</publisher-name>; <year>2018</year>. p. <fpage>23</fpage>&#x2013;<lpage>38</lpage>.</mixed-citation></ref>
<ref id="ref-125"><label>[125]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Gonz&#x00E1;lez-Ib&#x00E1;&#x00F1;ez</surname> <given-names>R</given-names></string-name>, <string-name><surname>Muresan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wacholder</surname> <given-names>N</given-names></string-name></person-group>. <article-title>Identifying sarcasm in Twitter: a closer look</article-title>. In: <conf-name>Proceedings of ACL-HLT; 2011 Jun 19&#x2013;24; Portland, OR, USA</conf-name>. p. <fpage>581</fpage>&#x2013;<lpage>6</lpage>.</mixed-citation></ref>
<ref id="ref-126"><label>[126]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Diaz</surname> <given-names>F</given-names></string-name>, <string-name><surname>Mitra</surname> <given-names>B</given-names></string-name>, <string-name><surname>Craswell</surname> <given-names>N</given-names></string-name></person-group>. <article-title>Query expansion with locally-trained word embeddings</article-title>. In: <conf-name>Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics; 2016 Aug 7&#x2013;12; Berlin, Germany</conf-name>. p. <fpage>367</fpage>&#x2013;<lpage>77</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/p16-1035</pub-id>.</mixed-citation></ref>
<ref id="ref-127"><label>[127]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Lazaridou</surname> <given-names>A</given-names></string-name>, <string-name><surname>Kuncoro</surname> <given-names>A</given-names></string-name>, <string-name><surname>Gribovskaya</surname> <given-names>E</given-names></string-name>, <string-name><surname>Agrawal</surname> <given-names>D</given-names></string-name>, <string-name><surname>Liska</surname> <given-names>A</given-names></string-name>, <string-name><surname>Terzi</surname> <given-names>T</given-names></string-name>, <etal>et al</etal></person-group>. <chapter-title>Mind the gap: assessing temporal generalization in neural language models</chapter-title>. In: <source>Advances in neural information processing systems (NeurIPS)</source>. Vol. <volume>34</volume>. <publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc.</publisher-name>; <year>2021</year>. p. <fpage>29348</fpage>&#x2013;<lpage>63</lpage>.</mixed-citation></ref>
<ref id="ref-128"><label>[128]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>McCloskey</surname> <given-names>M</given-names></string-name>, <string-name><surname>Cohen</surname> <given-names>NJ</given-names></string-name></person-group>. <chapter-title>Catastrophic interference in connectionist networks: the sequential learning problem</chapter-title>. In: <source>Psychology of learning and motivation</source>. <publisher-loc>Amsterdam, The Netherlands</publisher-loc>: <publisher-name>Elsevier</publisher-name>; <year>1989</year>. p. <fpage>109</fpage>&#x2013;<lpage>65</lpage>. doi:<pub-id pub-id-type="doi">10.1016/s0079-7421(08)60536-8</pub-id>.</mixed-citation></ref>
<ref id="ref-129"><label>[129]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Beltagy</surname> <given-names>I</given-names></string-name>, <string-name><surname>Peters</surname> <given-names>ME</given-names></string-name>, <string-name><surname>Cohan</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Longformer: the long-document transformer</article-title>. <comment>arXiv:200405150. 2020</comment>.</mixed-citation></ref>
<ref id="ref-130"><label>[130]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>He</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Feng</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Long document classification from local word glimpses via recurrent attention learning</article-title>. <source>IEEE Access</source>. <year>2019</year>;<volume>7</volume>:<fpage>40707</fpage>&#x2013;<lpage>18</lpage>. doi:<pub-id pub-id-type="doi">10.1109/access.2019.2907992</pub-id>.</mixed-citation></ref>
<ref id="ref-131"><label>[131]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Barnes</surname> <given-names>J</given-names></string-name>, <string-name><surname>Klinger</surname> <given-names>R</given-names></string-name>, <string-name><surname>Schulte im Walde</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Bilingual sentiment embeddings: joint projection of sentiment across languages</article-title>. In: <conf-name>Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics; 2018 Jul 15&#x2013;20; Melbourne, Australia</conf-name>. p. <fpage>2483</fpage>&#x2013;<lpage>93</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/p18-1231</pub-id>.</mixed-citation></ref>
<ref id="ref-132"><label>[132]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Sheng</surname> <given-names>E</given-names></string-name>, <string-name><surname>Chang</surname> <given-names>KW</given-names></string-name>, <string-name><surname>Natarajan</surname> <given-names>P</given-names></string-name>, <string-name><surname>Peng</surname> <given-names>N</given-names></string-name></person-group>. <article-title>The woman worked as a babysitter: on biases in language generation</article-title>. In: <conf-name>Proceedings of EMNLP-IJCNLP; 2019 Nov 3&#x2013;7; Hong Kong, China</conf-name>. p. <fpage>3407</fpage>&#x2013;<lpage>12</lpage>.</mixed-citation></ref>
<ref id="ref-133"><label>[133]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Kiritchenko</surname> <given-names>S</given-names></string-name>, <string-name><surname>Mohammad</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Examining gender and race bias in two hundred sentiment analysis systems</article-title>. In: <conf-name>Proceedings of the Seventh Joint Conference on Lexical and Computational Semantics; 2018 Jun 5&#x2013;6; New Orleans, LA, USA</conf-name>. p. <fpage>43</fpage>&#x2013;<lpage>53</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/s18-2005</pub-id>.</mixed-citation></ref>
<ref id="ref-134"><label>[134]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>EJ</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wallis</surname> <given-names>P</given-names></string-name>, <string-name><surname>Allen-Zhu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>LoRA: low-rank adaptation of large language models</article-title>. In: <conf-name>Proceedings of the International Conference on Learning Representations (ICLR); 2022 Apr 25&#x2013;29; Online</conf-name>.</mixed-citation></ref>
<ref id="ref-135"><label>[135]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Dettmers</surname> <given-names>T</given-names></string-name>, <string-name><surname>Holtzman</surname> <given-names>A</given-names></string-name>, <string-name><surname>Pagnoni</surname> <given-names>A</given-names></string-name>, <string-name><surname>Zettlemoyer</surname> <given-names>L</given-names></string-name></person-group>. <article-title>QLoRA: efficient finetuning of quantized LLMs</article-title>. In: <conf-name>Proceedings of the NIPS&#x2019;23: Proceedings of the 37th International Conference on Neural Information Processing Systems; 2023 Dec 10&#x2013;16; New Orleans, LA, USA</conf-name>. p. <fpage>10088</fpage>&#x2013;<lpage>115</lpage>. doi:<pub-id pub-id-type="doi">10.52202/075280-0441</pub-id>.</mixed-citation></ref>
<ref id="ref-136"><label>[136]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Danilevsky</surname> <given-names>M</given-names></string-name>, <string-name><surname>Qian</surname> <given-names>K</given-names></string-name>, <string-name><surname>Aharonov</surname> <given-names>R</given-names></string-name>, <string-name><surname>Katsis</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Kawas</surname> <given-names>B</given-names></string-name>, <string-name><surname>Sen</surname> <given-names>P</given-names></string-name></person-group>. <article-title>A survey of the state of explainable AI for natural language processing</article-title>. In: <conf-name>Proceedings of the 1st Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics and the 10th International Joint Conference on Natural Language Processing; 2020 Dec 4&#x2013;7; Suzhou, China</conf-name>. p. <fpage>447</fpage>&#x2013;<lpage>59</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2020.aacl-main.46</pub-id>.</mixed-citation></ref>
<ref id="ref-137"><label>[137]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ribeiro</surname> <given-names>MT</given-names></string-name>, <string-name><surname>Singh</surname> <given-names>S</given-names></string-name>, <string-name><surname>Guestrin</surname> <given-names>C</given-names></string-name></person-group>. <article-title>&#x0201C;Why should I trust you?&#x201D;: explaining the predictions of any classifier</article-title>. In: <conf-name>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining; 2016 Aug 13&#x2013;17; San Francisco, CA, USA</conf-name>. p. <fpage>1135</fpage>&#x2013;<lpage>44</lpage>. doi:<pub-id pub-id-type="doi">10.1145/2939672.2939778</pub-id>.</mixed-citation></ref>
<ref id="ref-138"><label>[138]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Rajani</surname> <given-names>NF</given-names></string-name>, <string-name><surname>McCann</surname> <given-names>B</given-names></string-name>, <string-name><surname>Xiong</surname> <given-names>C</given-names></string-name>, <string-name><surname>Socher</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Explain yourself! Leveraging language models for commonsense reasoning</article-title>. In: <conf-name>Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics; 2019 Jul 28&#x2013;Aug 2; Florence, Italy</conf-name>. p. <fpage>4932</fpage>&#x2013;<lpage>42</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/p19-1487</pub-id>.</mixed-citation></ref>
<ref id="ref-139"><label>[139]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yin</surname> <given-names>W</given-names></string-name>, <string-name><surname>Hay</surname> <given-names>J</given-names></string-name>, <string-name><surname>Roth</surname> <given-names>D</given-names></string-name></person-group>. <article-title>Benchmarking zero-shot text classification: datasets, evaluation and entailment approach</article-title>. In: <conf-name>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP); 2019 Nov 3&#x2013;7; Hong Kong, China</conf-name>. p. <fpage>3914</fpage>&#x2013;<lpage>23</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/d19-1404</pub-id>.</mixed-citation></ref>
</ref-list>
</back></article>