<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="review-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMES</journal-id>
<journal-id journal-id-type="nlm-ta">CMES</journal-id>
<journal-id journal-id-type="publisher-id">CMES</journal-id>
<journal-title-group>
<journal-title>Computer Modeling in Engineering &#x0026; Sciences</journal-title>
</journal-title-group>
<issn pub-type="epub">1526-1506</issn>
<issn pub-type="ppub">1526-1492</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">83586</article-id>
<article-id pub-id-type="doi">10.32604/cmes.2026.083586</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Review</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>A Comprehensive Review of Complex Logical Reasoning in Large Vision-Language Models</article-title>
<alt-title alt-title-type="left-running-head">A Comprehensive Review of Complex Logical Reasoning in Large Vision-Language Models</alt-title>
<alt-title alt-title-type="right-running-head">A Comprehensive Review of Complex Logical Reasoning in Large Vision-Language Models</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Jin</surname><given-names>Weiqiang</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref><xref ref-type="author-notes" rid="afn1">#</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Liu</surname><given-names>Yang</given-names></name><xref ref-type="aff" rid="aff-2">2</xref><xref ref-type="author-notes" rid="afn1">#</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Gao</surname><given-names>Yang</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="author-notes" rid="afn1">#</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Tang</surname><given-names>Shixiang</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Zhou</surname><given-names>Yanghao</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Qi</surname><given-names>Jinhu</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Zhang</surname><given-names>Wentao</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western"><surname>Wang</surname><given-names>Junli</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-9" contrib-type="author">
<name name-style="western"><surname>Gao</surname><given-names>Jing</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-10" contrib-type="author">
<name name-style="western"><surname>Ma</surname><given-names>Yue</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-11" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Zhang</surname><given-names>Ziwei</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><email>ziwei.zhang@xjtu.edu.cn</email></contrib>
<contrib id="author-12" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Zhao</surname><given-names>Biao</given-names></name><xref ref-type="aff" rid="aff-2">2</xref><email>biaozhao@xjtu.edu.cn</email></contrib>
<aff id="aff-1"><label>1</label><institution>Institute of SRIICL, Xi&#x2019;an Jiaotong University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country></aff>
<aff id="aff-2"><label>2</label><institution>School of Information and Communications Engineering, Xi&#x2019;an Jiaotong University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country></aff>
<aff id="aff-3"><label>3</label><institution>Department of Electrical and Computer Engineering, National University of Singapore</institution>, <addr-line>Kent Ridge</addr-line>, <country>Singapore</country></aff>
<aff id="aff-4"><label>4</label><institution>Department of Computer Science and Engineering, The Chinese University of Hong Kong</institution>, <addr-line>Hong Kong SAR</addr-line>, <country>China</country></aff>
<aff id="aff-5"><label>5</label><institution>School of Computer Science and Technology, University of Science and Technology of China</institution>, <addr-line>Hefei</addr-line>, <country>China</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Authors: Ziwei Zhang. Email: <email>ziwei.zhang@xjtu.edu.cn</email>; Biao Zhao. Email: <email>biaozhao@xjtu.edu.cn</email></corresp>
<fn id="afn1">
<p><sup>#</sup>These authors contributed equally to this work</p>
</fn>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2026</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>27</day><month>07</month><year>2026</year>
</pub-date>
<volume>148</volume>
<issue>1</issue>
<elocation-id>1</elocation-id>
<history>
<date date-type="received">
<day>10</day>
<month>04</month>
<year>2026</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>06</month>
<year>2026</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2026 The Authors. Published by Tech Science Press.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>The Authors</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMES_83586.pdf"></self-uri>
<abstract>
<p>Large Vision-Language Models (LVLMs) have achieved strong performance in multimodal perception, understanding, and generation, but their ability to perform complex logical reasoning remains insufficiently understood. In particular, it is still unclear whether current LVLMs can reliably conduct explicit logical operations, multi-step inference, abstract relational reasoning, and cross-modal evidence integration. Reasoning abilities such as deductive, inductive, abductive, multi-hop, and causal inference are fundamental to robust decision making, trustworthy interaction, and real-world deployment, yet they have not been systematically examined in the LVLM literature. Existing surveys mainly discuss mathematical reasoning, general multimodal intelligence, or benchmark progress, but they do not provide a unified account of complex logical reasoning in LVLMs, including its definition, reasoning types, modeling paradigms, evaluation protocols, and unresolved limitations. To address this gap, this survey develops a unified analytical framework for complex logical reasoning in LVLMs. This survey provides a structured review of this emerging area. We first formalize complex logical reasoning in multimodal settings and organize the literature into five recurrent reasoning families: deductive, inductive, abductive, multi-hop, and causal reasoning. We then review reasoning-oriented LVLM architectures, including unified, modular, and tool-augmented paradigms, and summarize major reasoning mechanisms such as chain-of-thought, program-based reasoning, self-correction, and interpretability-oriented analysis. We further examine representative benchmarks and evaluation protocols, with particular attention to the mismatch between final-answer accuracy and genuine reasoning validity. Based on empirical evidence from representative LVLMs and datasets, we identify common capability trends, recurring failure modes, and key open challenges. Our analysis shows that current LVLMs still struggle with reasoning faithfulness, long-horizon inference, cross-modal grounding, hallucination control, and process-aware evaluation. Finally, we outline future directions in reasoning-oriented data construction, model design, training strategies, evaluation methodology, and deployment. Overall, this survey offers a unified conceptual framework and technical roadmap for advancing LVLMs from strong perceptual systems toward reliable multimodal reasoning agents.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Large vision-language model</kwd>
<kwd>complex logical reasoning</kwd>
<kwd>multimodal reasoning</kwd>
<kwd>chain-of-thought</kwd>
<kwd>evaluation benchmark</kwd>
<kwd>reasoning faithfulness</kwd>
</kwd-group></article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>In recent years, artificial intelligence has witnessed rapid progress in large language models (LLMs) and their multimodal extensions [<xref ref-type="bibr" rid="ref-1">1</xref>&#x2013;<xref ref-type="bibr" rid="ref-3">3</xref>]. Large Vision-Language Models (LVLMs) [<xref ref-type="bibr" rid="ref-4">4</xref>], which integrate visual perception with language understanding and generation, have demonstrated unprecedented performance in various multimodal tasks such as image captioning, image-text dialogue, and visual question answering (VQA) [<xref ref-type="bibr" rid="ref-5">5</xref>&#x2013;<xref ref-type="bibr" rid="ref-8">8</xref>]. <xref ref-type="fig" rid="fig-1">Fig. 1</xref> provides an overview of this transition from multimodal perception to complex logical reasoning and summarizes the survey roadmap.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>From multimodal perception to complex logical reasoning in LVLMs: an overview of the capability transition, the scope of this survey, and the roadmap of our review.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-1.tif"/>
</fig>
<p>Complex logical reasoning is essential for moving LVLMs beyond perception-oriented tasks toward reliable multimodal agents [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-2">2</xref>]. In practical settings, models must often infer hidden relations, combine distributed evidence, explain causal events, resolve inconsistencies between images and text, and justify why a conclusion follows from the available evidence. The same challenge extends to dynamic multimodal scenarios, where event ordering or audiovisual cues may also need to be interpreted coherently [<xref ref-type="bibr" rid="ref-9">9</xref>,<xref ref-type="bibr" rid="ref-10">10</xref>].</p>
<p>Accordingly, the central problem addressed in this survey is the gap between LVLMs&#x2019; strong multimodal perception and their still limited ability to perform reliable complex logical reasoning. In this paper, we use the term to denote multimodal inference that requires explicit transformation of premises, maintenance of multi-step dependencies, or composition of evidence across sources, rather than direct perceptual lookup or simple retrieval. While modern LVLMs excel at perceptual tasks such as object recognition and scene description, this limitation is especially concerning because reasoning quality directly affects high-stakes applications such as medical diagnosis, legal analysis, scientific discovery, and complex decision-making [<xref ref-type="bibr" rid="ref-11">11</xref>&#x2013;<xref ref-type="bibr" rid="ref-13">13</xref>].</p>
<p>In medical imaging, for example, LVLMs may need to combine visual findings with textual clinical context; in autonomous driving, they must connect object states, scene dynamics, and safety constraints before acting [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-14">14</xref>].</p>
<p>Despite rapid progress, research on complex logical reasoning in LVLMs remains fragmented [<xref ref-type="bibr" rid="ref-2">2</xref>,<xref ref-type="bibr" rid="ref-3">3</xref>,<xref ref-type="bibr" rid="ref-15">15</xref>]. Previous reviews have discussed mathematical reasoning or broad multimodal capability, but they do not provide a focused treatment of logical reasoning as a distinct multimodal research problem [<xref ref-type="bibr" rid="ref-16">16</xref>&#x2013;<xref ref-type="bibr" rid="ref-20">20</xref>]. As a result, three gaps remain especially prominent: the concept itself is still ambiguously defined; current benchmarks emphasize final-answer accuracy more than reasoning faithfulness; and advances in architectures, prompting, and evaluation are often reported in isolated settings that are difficult to compare systematically [<xref ref-type="bibr" rid="ref-21">21</xref>&#x2013;<xref ref-type="bibr" rid="ref-24">24</xref>]. Accordingly, the survey is organized in a problem-to-evidence order: we first operationalize the target reasoning scope, then compare architectures and reasoning mechanisms under a common framework, and finally synthesize benchmark limitations, empirical trends, and unresolved bottlenecks.</p>
<sec id="s1_1">
<label>1.1</label>
<title>Research Motivation</title>
<p>This survey is motivated by three specific concerns, detailed below:</p>
<p>(1) The definition of <italic>complex logical reasoning</italic> in multimodal settings is often ambiguous, with unclear boundaries relative to perceptual understanding, commonsense reasoning, and mathematical problem solving [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>,<xref ref-type="bibr" rid="ref-26">26</xref>].</p>
<p>(2) Meanwhile, existing research lacks a unified evaluation framework. Current benchmarks and methodologies are scattered and typically focus on final-answer accuracy, such as OlympiadBench [<xref ref-type="bibr" rid="ref-27">27</xref>] and TheoremQA [<xref ref-type="bibr" rid="ref-28">28</xref>], offering limited insight into reasoning faithfulness and intermediate inference processes [<xref ref-type="bibr" rid="ref-29">29</xref>&#x2013;<xref ref-type="bibr" rid="ref-31">31</xref>].</p>
<p>(3) In addition, this field is rapidly evolving, with architectural innovations, prompting strategies, reasoning mechanisms, and evaluation protocols emerging at an accelerating pace [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-32">32</xref>]. However, these techniques are frequently assessed in isolated settings, making systematic comparison across methods difficult [<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-29">29</xref>].</p>
<p>To address these issues, this survey focuses specifically on <italic>complex logical reasoning</italic> in LVLMs, distinguishing it from related areas such as mathematical reasoning, commonsense reasoning, and visual perception [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-26">26</xref>,<xref ref-type="bibr" rid="ref-33">33</xref>]. We define it as multimodal reasoning in which the final answer cannot be obtained by direct perceptual lookup alone, but instead requires at least one inferential transformation such as explicit logical operations, multi-step dependency tracking, abstract relation handling, or cross-source evidence composition [<xref ref-type="bibr" rid="ref-10">10</xref>,<xref ref-type="bibr" rid="ref-23">23</xref>,<xref ref-type="bibr" rid="ref-34">34</xref>&#x2013;<xref ref-type="bibr" rid="ref-36">36</xref>]. We primarily cover work published from 2022 to early 2026, while retaining foundational earlier studies for historical context [<xref ref-type="bibr" rid="ref-22">22</xref>,<xref ref-type="bibr" rid="ref-33">33</xref>,<xref ref-type="bibr" rid="ref-37">37</xref>&#x2013;<xref ref-type="bibr" rid="ref-39">39</xref>]. Our survey covers five primary dimensions: conceptual foundations, architectural paradigms, reasoning mechanisms, evaluation benchmarks, and empirical findings, along with discussions of current challenges and future directions [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>].</p>
<p>The main contributions of this survey are summarized as follows:<list list-type="simple">
<list-item>
<label>1.</label>
<p><bold>Problem formalization and conceptual taxonomy.</bold> We clarify the scope of complex logical reasoning in LVLMs and distinguish it from visual perception, commonsense reasoning, mathematical reasoning, and factual recall. We further organize existing studies into five recurrent reasoning families under a hierarchical taxonomy: deductive, inductive, abductive, multi-hop, and causal reasoning [<xref ref-type="bibr" rid="ref-10">10</xref>,<xref ref-type="bibr" rid="ref-21">21</xref>,<xref ref-type="bibr" rid="ref-23">23</xref>,<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>].</p></list-item>
<list-item>
<label>2.</label>
<p><bold>Structured methodological analysis.</bold> We analyze reasoning-oriented LVLM architectures, including unified, modular, and tool-augmented paradigms [<xref ref-type="bibr" rid="ref-40">40</xref>&#x2013;<xref ref-type="bibr" rid="ref-42">42</xref>], and review major reasoning mechanisms such as chain-of-thought, program-based reasoning, self-correction, and interpretability-oriented analysis [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-43">43</xref>,<xref ref-type="bibr" rid="ref-44">44</xref>]. For representative methods, we highlight their methodological focus, typical application scenarios, advantages, and limitations to facilitate comparison across different lines of work.</p></list-item>
<list-item>
<label>3.</label>
<p><bold>Critical evaluation analysis.</bold> We present a comprehensive overview of benchmarks and evaluation protocols for complex logical reasoning in LVLMs, including existing datasets, evaluation metrics, and their limitations [<xref ref-type="bibr" rid="ref-30">30</xref>,<xref ref-type="bibr" rid="ref-31">31</xref>,<xref ref-type="bibr" rid="ref-45">45</xref>]. We compare the characteristics of different benchmarks, highlighting gaps in current evaluation methodologies and opportunities for future benchmark development [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-46">46</xref>]. In particular, we summarize commonly used evaluation criteria, such as answer accuracy, reasoning consistency, robustness, and process-level faithfulness, so that readers can better understand the utility and limitations of different datasets and metrics.</p></list-item>
<list-item>
<label>4.</label>
<p><bold>Empirical analysis.</bold> We synthesize representative empirical findings to reveal recurring behavior patterns, capability boundaries, and common failure modes [<xref ref-type="bibr" rid="ref-47">47</xref>&#x2013;<xref ref-type="bibr" rid="ref-49">49</xref>]. We further analyze factors that contribute to reasoning success and failure to guide future developments [<xref ref-type="bibr" rid="ref-25">25</xref>,<xref ref-type="bibr" rid="ref-49">49</xref>].</p></list-item>
<list-item>
<label>5.</label>
<p><bold>Challenges and forward-looking research roadmap.</bold> We identify several major challenges hindering the advancement of complex logical reasoning in LVLMs and propose future research directions across various dimensions [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-24">24</xref>].</p></list-item>
</list></p>
<p>In addition to the above contributions, this survey also provides extensive references to foundational and state-of-the-art works, serving as a comprehensive resource for researchers entering the field [<xref ref-type="bibr" rid="ref-2">2</xref>,<xref ref-type="bibr" rid="ref-3">3</xref>,<xref ref-type="bibr" rid="ref-50">50</xref>]. We organize our discussion around clear conceptual distinctions and provide examples drawn from recent literature to illustrate key points [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>].</p>
</sec>
<sec id="s1_2">
<label>1.2</label>
<title>Literature Search and Screening Methodology</title>
<p>We conducted a structured search across major academic and open-access platforms, including Google Scholar<xref ref-type="fn" rid="fn-1"><sup>1</sup></xref><fn id="fn-1"><label>1</label><p>Google Scholar: <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com">https://scholar.google.com</ext-link></p></fn>, arXiv<xref ref-type="fn" rid="fn-2"><sup>2</sup></xref><fn id="fn-2"><label>2</label><p>ArXiv Platform: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org">https://arxiv.org</ext-link></p></fn>, IEEE Xplore<xref ref-type="fn" rid="fn-3"><sup>3</sup></xref><fn id="fn-3"><label>3</label><p>IEEE Xplore: <ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org">https://ieeexplore.ieee.org</ext-link></p></fn>, ACM Digital Library<xref ref-type="fn" rid="fn-4"><sup>4</sup></xref><fn id="fn-4"><label>4</label><p>ACM Digital Library: <ext-link ext-link-type="uri" xlink:href="https://dl.acm.org">https://dl.acm.org</ext-link></p></fn>, SpringerLink<xref ref-type="fn" rid="fn-5"><sup>5</sup></xref><fn id="fn-5"><label>5</label><p>SpringerLink: <ext-link ext-link-type="uri" xlink:href="https://link.springer.com">https://link.springer.com</ext-link></p></fn>, ScienceDirect<xref ref-type="fn" rid="fn-6"><sup>6</sup></xref><fn id="fn-6"><label>6</label><p>ScienceDirect: <ext-link ext-link-type="uri" xlink:href="https://www.sciencedirect.com">https://www.sciencedirect.com</ext-link></p></fn>, OpenReview<xref ref-type="fn" rid="fn-7"><sup>7</sup></xref><fn id="fn-7"><label>7</label><p>OpenReview: <ext-link ext-link-type="uri" xlink:href="https://openreview.net">https://openreview.net</ext-link></p></fn>, and the ACL Anthology<xref ref-type="fn" rid="fn-8"><sup>8</sup></xref><fn id="fn-8"><label>8</label><p>ACL Anthology: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org">https://aclanthology.org</ext-link></p></fn>. To improve coverage, we also performed backward snowballing from key survey and benchmark papers. The literature search for the present revision was last updated on 30 May 2026. The search strategy combined keywords related to LVLMs and reasoning, such as &#x201C;large vision-language model&#x201D;, &#x201C;multimodal reasoning&#x201D;, &#x201C;logical reasoning&#x201D;, &#x201C;chain-of-thought&#x201D;, and specific reasoning types (e.g., deductive, inductive, causal). These were used both individually and in combination (e.g., &#x201C;LVLM AND logical reasoning&#x201D;). We applied the following inclusion criteria: (1) studies on LVLMs or multimodal reasoning systems; (2) work addressing reasoning mechanisms, architectures, benchmarks, evaluation, or empirical failure analysis; (3) peer-reviewed papers or publicly available preprints with sufficient technical detail; and (4) clear relevance to complex logical reasoning. We focused primarily on works published between 2022 and early 2026, while retaining a limited number of earlier foundational studies. Exclusion criteria included duplicate records, unimodal papers without direct multimodal transfer relevance, multimodal papers focused only on captioning or retrieval without reasoning analysis, leaderboard-style reports lacking methodological detail, and items for which the full technical content could not be examined.</p>
<p>The screening process consisted of three stages: (1) initial collection via database search and reference expansion; (2) title and abstract filtering using the above inclusion/exclusion rules; and (3) full-text review for technical relevance, contribution type, and fit to one or more survey dimensions. The retained studies were then organized into conceptual/foundational works, architecture-oriented works, reasoning-mechanism works, and benchmark/evaluation works. Because this survey was developed as a structured narrative review rather than a preregistered systematic review, and because the earliest candidate pool was expanded iteratively during drafting and backward snowballing, exact stage-wise exclusion rates were not preserved in a form suitable for retrospective reporting. We therefore report the final corpus composition as a transparency summary rather than claiming PRISMA-style stage counts for the earliest drafting stage.</p>
<p>Following this process, the retained corpus comprises 138 final cited references and spans international conferences, journal publications, and a large proportion of preprints, reflecting the fast-evolving nature of the field. Most works were published between 2023 and 2025.</p>
<p>For greater quantitative transparency, we further summarize the retained corpus here. The current manuscript cites 138 unique works in its final reference list. The cited corpus is strongly recent: 101 of the 138 cited works (73.2%) were published between 2023 and 2025. To summarize topical coverage without forcing each paper into a single exclusive bin, we report overlapping section-level counts: 58 architecture-related citations, 42 reasoning-mechanism citations, and 56 benchmark/evaluation citations, in addition to conceptual and survey papers used for problem framing. This composition improves coverage of a rapidly moving field, but it also introduces clear coverage biases: the survey favors recent English-language public papers, benchmark-heavy studies, and open preprints, while closed industrial evaluations, negative results, and papers with limited methodological disclosure are more likely to be underrepresented.</p>
</sec>
<sec id="s1_3">
<label>1.3</label>
<title>Content Organization</title>
<p>The remainder of this survey is organized as follows. <xref ref-type="sec" rid="s2">Section 2</xref> introduces conceptual foundations of complex logical reasoning in LVLMs and presents the proposed taxonomy. <xref ref-type="sec" rid="s3">Section 3</xref> examines LVLM architectures designed for logical reasoning. <xref ref-type="sec" rid="s4">Section 4</xref> reviews reasoning mechanisms and strategies. <xref ref-type="sec" rid="s5">Section 5</xref> summarizes benchmarks and evaluation protocols <xref ref-type="sec" rid="s6">Section 6</xref> discusses challenges and open problems. <xref ref-type="sec" rid="s7">Section 7</xref> outlines future research directions. <xref ref-type="sec" rid="s8">Section 8</xref> concludes the survey.</p>
</sec>
</sec>
<sec id="s2">
<label>2</label>
<title>Conceptual Foundations: Complex Logical Reasoning in LVLMs</title>
<p>Understanding complex logical reasoning in the context of LVLMs requires precise definitions and clear distinctions from related concepts. This section establishes the conceptual foundations by defining what constitutes logical reasoning in multimodal contexts and presenting a comprehensive taxonomy of reasoning types. Our aim is to provide a rigorous conceptual framework that can guide both research and evaluation in this rapidly evolving field [<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>].</p>
<p>The study of complex logical reasoning in LVLMs draws upon multiple disciplines, including philosophy (particularly logic and epistemology), cognitive science (studying human reasoning processes), and artificial intelligence (formal methods for automated inference). Furthermore, Zhou et al. [<xref ref-type="bibr" rid="ref-19">19</xref>] suggest that reasoning performance depends not only on model architecture but also on training strategy, reward design, and trajectory optimization. By integrating perspectives from these fields, we can develop a more complete understanding of what complex logical reasoning means in multimodal contexts and what challenges must be addressed to advance LVLM capabilities [<xref ref-type="bibr" rid="ref-21">21</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>].</p>
<sec id="s2_1">
<label>2.1</label>
<title>Defining Complex Logical Reasoning in Multimodal Contexts</title>
<p>Complex logical reasoning in LVLMs differs fundamentally from pure textual reasoning and simple visual perception. The core distinction lies between perceptual-level and logical-reasoning-level understanding. Perceptual-level understanding involves recognizing objects, identifying attributes, and describing scenes, tasks that modern LVLMs perform with remarkable accuracy [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-40">40</xref>,<xref ref-type="bibr" rid="ref-52">52</xref>,<xref ref-type="bibr" rid="ref-53">53</xref>]. In contrast, complex logical reasoning requires manipulating abstract representations, applying inference rules, and deriving conclusions not explicitly given in the input [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-21">21</xref>,<xref ref-type="bibr" rid="ref-23">23</xref>]. <xref ref-type="fig" rid="fig-2">Fig. 2</xref> illustrates this transition from perceptual recognition to higher-level multimodal inference.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>From perceptual understanding to complex logical reasoning in multimodal contexts. Multimodal inputs are first processed at the perceptual level to recognize objects, attributes, and basic relations, and are then transformed into higher-level reasoning representations that support multi-step inference and implicit conclusion generation.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-2.tif"/>
</fig>
<p>The distinction between perceptual and complex logical reasoning can also be understood through the lens of cognitive hierarchies. Perceptual cognition operates on sensory data to extract features, objects, and basic relations. Logical cognition operates on these extracted representations to reason about implications, dependencies, and abstract relationships. While perception is often relatively direct for both humans and AI systems, complex logical reasoning typically requires deliberate multi-step inference and may rely on explicit or implicit logical rules [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>].</p>
<p>Complex logical reasoning in LVLM contexts exhibits several key characteristics that distinguish it from simpler forms of multimodal understanding:<list list-type="simple">
<list-item>
<label>1.</label>
<p><bold>Multi-Step Nature</bold>: Complex logical reasoning typically requires chaining multiple inference steps, where each step builds upon previous conclusions. This property distinguishes logical reasoning from single-hop inference, where conclusions can be derived directly from premises without intermediate steps. Multi-step reasoning is essential for tasks such as solving complex puzzles, analyzing scenarios with multiple constraints, and answering questions that require integrating information from different sources [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-54">54</xref>&#x2013;<xref ref-type="bibr" rid="ref-57">57</xref>].</p></list-item>
<list-item>
<label>2.</label>
<p><bold>Combinatorial Mathematicality</bold>: Reasoning often involves combining information from multiple sources or modalities, requiring the integration of visual evidence with textual premises. This combinatorial aspect is particularly challenging because it requires maintaining consistency across different representational formats and ensuring that conclusions properly account for all available evidence. For example, a reasoning task may require combining visual evidence about object positions with textual premises about their relationships in order to infer their interactions [<xref ref-type="bibr" rid="ref-36">36</xref>,<xref ref-type="bibr" rid="ref-58">58</xref>].</p></list-item>
<list-item>
<label>3.</label>
<p><bold>Abstraction</bold>: Complex logical reasoning operates on abstract representations rather than concrete perceptual features, involving concepts such as relations, dependencies, and constraints. Abstraction allows reasoning to generalize across specific instances and apply general rules to novel situations. In LVLMs, this requires moving beyond pixel-level representations toward more abstract representations that capture the semantic and logical structure of scenes [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-22">22</xref>,<xref ref-type="bibr" rid="ref-46">46</xref>].</p></list-item>
<list-item>
<label>4.</label>
<p><bold>Cross-Modal Consistency</bold>: In multimodal settings, complex logical reasoning requires maintaining consistency across visual and linguistic representations, ensuring that conclusions drawn from visual evidence align with textual information. This requirement is particularly challenging because different modalities may provide complementary, partially overlapping, or even conflicting information. Reasoning systems must therefore detect and resolve such conflicts to produce valid conclusions [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-59">59</xref>&#x2013;<xref ref-type="bibr" rid="ref-61">61</xref>].</p></list-item>
</list></p>
<p>Operationally, this survey treats a task as <italic>complex logical reasoning</italic> only when the final answer cannot be obtained through a single perceptual lookup or plain evidence retrieval, but instead requires at least one non-trivial inferential transformation over multimodal evidence. Such transformation may take the form of rule application, hypothesis selection, causal attribution, constraint satisfaction, or cross-source composition. Under this criterion, multi-step length alone is not sufficient: a retrieval pipeline that merely gathers several relevant facts without comparing, composing, or logically constraining them is not counted as complex logical reasoning in our taxonomy. By contrast, a multi-hop setting falls within scope when intermediate evidence must be integrated into a new conclusion whose validity depends on the consistency of the whole inference chain [<xref ref-type="bibr" rid="ref-54">54</xref>&#x2013;<xref ref-type="bibr" rid="ref-56">56</xref>].</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Main Types of Complex Logical Reasoning in LVLM Research</title>
<p>Drawing upon established frameworks from cognitive science and logic, we organize the literature into five recurrent reasoning families: deductive, inductive, abductive, multi-hop, and causal reasoning [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>]. We do not treat these families as a strictly flat or mutually exclusive partition. Instead, we use a hierarchical reading of the taxonomy: deductive, inductive, abductive, and causal reasoning are <italic>operator-centered</italic> categories defined by the dominant inferential goal, whereas multi-hop reasoning is a <italic>structural</italic> category indicating that evidence or sub-goals must be chained across several steps. This framing reflects actual LVLM research more faithfully, because a model may be abductive and multi-hop at the same time, or causal and multi-hop, without creating a contradiction in the classification [<xref ref-type="bibr" rid="ref-24">24</xref>]. <xref ref-type="fig" rid="fig-3">Fig. 3</xref> summarizes this hierarchical organization of reasoning families.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Taxonomy of the main reasoning families studied in LVLM research. Deductive, inductive, abductive, and causal reasoning correspond to dominant inference operators, whereas multi-hop reasoning denotes a cross-cutting structural pattern in which evidence must be chained across steps.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-3.tif"/>
</fig>
<p>To improve the comparability of related studies, <xref ref-type="table" rid="table-1">Table 1</xref> links each family to its operational definition and typical task settings. We assign representative works according to their <italic>dominant inference objective</italic>: formal entailment for deductive reasoning, rule abstraction for inductive reasoning, explanation generation for abductive reasoning, intervention or cause-effect analysis for causal reasoning, and cross-step evidence chaining for multi-hop reasoning. When a work clearly spans multiple families, we use its main evaluation target as the primary label and discuss the overlap explicitly in the surrounding text.<xref ref-type="fn" rid="fn-9"><sup>9</sup></xref><fn id="fn-9"><label>9</label><p>The representative works are categorized by their dominant inference objective rather than by an assumption of mutual exclusivity. In particular, multi-hop reasoning is treated as a cross-cutting structural pattern that may co-occur with deductive, abductive, or causal reasoning.</p></fn> The subsequent architecture, mechanism, and benchmark sections further discuss the methodological strengths, application scenarios, evaluation metrics, and limitations of these representative works.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Hierarchical taxonomy of complex logical reasoning in LVLM research. Deductive, inductive, abductive, and causal reasoning are operator-centered families, while multi-hop reasoning is a structural family. Representative works are categorized by their dominant inference objective.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>Reasoning Type</th>
<th>Definition</th>
<th>Typical Tasks</th>
<th>Representative Works</th>
</tr>
</thead>
<tbody>
<tr>
<td>Deductive Reasoning</td>
<td>Deriving specific conclusions from general premises through formal logic rules</td>
<td>Rule application, premise validation, logical entailment</td>
<td>LINC [<xref ref-type="bibr" rid="ref-62">62</xref>], NS-CL [<xref ref-type="bibr" rid="ref-63">63</xref>], ReClor [<xref ref-type="bibr" rid="ref-64">64</xref>], R1-Onevision [<xref ref-type="bibr" rid="ref-65">65</xref>]</td>
</tr>
<tr>
<td>Inductive Reasoning</td>
<td>Abstracting general patterns or rules from specific instances</td>
<td>Pattern discovery, rule induction, generalization</td>
<td>RAVEN [<xref ref-type="bibr" rid="ref-22">22</xref>], IndMKG [<xref ref-type="bibr" rid="ref-66">66</xref>], InPhyRe [<xref ref-type="bibr" rid="ref-67">67</xref>], WeThink [<xref ref-type="bibr" rid="ref-68">68</xref>], Molmo [<xref ref-type="bibr" rid="ref-69">69</xref>]</td>
</tr>
<tr>
<td>Abductive Reasoning</td>
<td>Inferring the most plausible explanation for observed phenomena</td>
<td>Cause inference, explanation generation, diagnosis</td>
<td>Reasoner [<xref ref-type="bibr" rid="ref-23">23</xref>], MAR [<xref ref-type="bibr" rid="ref-9">9</xref>], Vision-R1 [<xref ref-type="bibr" rid="ref-70">70</xref>], WeThink [<xref ref-type="bibr" rid="ref-68">68</xref>]</td>
</tr>
<tr>
<td>Multi-hop Reasoning</td>
<td>Chaining evidence or sub-goals across multiple steps; may co-occur with deductive, abductive, or causal operators</td>
<td>Composite question answering, cross-modality inference</td>
<td>BRIDGE [<xref ref-type="bibr" rid="ref-55">55</xref>], M3GQA [<xref ref-type="bibr" rid="ref-54">54</xref>], MultiHop-RAG [<xref ref-type="bibr" rid="ref-56">56</xref>], Chameleon [<xref ref-type="bibr" rid="ref-71">71</xref>], LLaVA-NeXT-Interleave [<xref ref-type="bibr" rid="ref-72">72</xref>], Qwen2.5-VL [<xref ref-type="bibr" rid="ref-73">73</xref>]</td>
</tr>
<tr>
<td>Causal Reasoning</td>
<td>Identifying causal relationships between events or variables</td>
<td>Causal attribution, effect prediction, intervention analysis</td>
<td>CausalVLBench [<xref ref-type="bibr" rid="ref-34">34</xref>], CausalBench [<xref ref-type="bibr" rid="ref-35">35</xref>], MuCR [<xref ref-type="bibr" rid="ref-10">10</xref>], VcCoT [<xref ref-type="bibr" rid="ref-10">10</xref>], Reasoner [<xref ref-type="bibr" rid="ref-23">23</xref>], Vision-R1 [<xref ref-type="bibr" rid="ref-70">70</xref>]</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><bold>Deductive Reasoning</bold>. Deductive reasoning involves deriving specific conclusions from general premises through the application of formal logical rules [<xref ref-type="bibr" rid="ref-21">21</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>]. In LVLM contexts, deductive reasoning typically appears in tasks requiring rule application, premise validation, or logical entailment determination [<xref ref-type="bibr" rid="ref-64">64</xref>]. For example, given a set of premises about relationships between objects and a query about a specific relationship, deductive reasoning enables the model to infer the answer by applying the relevant rule structure.</p>
<p>The formal nature of deductive reasoning makes it particularly amenable to neuro-symbolic approaches that combine neural representation learning with symbolic inference engines [<xref ref-type="bibr" rid="ref-41">41</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>]. Works such as ReClor [<xref ref-type="bibr" rid="ref-64">64</xref>] have developed benchmarks specifically targeting deductive reasoning capabilities in language models. ReClor presents reading comprehension questions that require complex logical reasoning to determine entailment or contradiction relationships between premises and conclusions.</p>
<p>A key challenge in deductive reasoning for LVLMs lies in handling the semantic gap between natural language premises and formal logical representations [<xref ref-type="bibr" rid="ref-62">62</xref>]. Natural language is often ambiguous and may not map cleanly to formal logical forms. Furthermore, real-world premises may be incomplete or may contain implicit assumptions that must be identified for valid deduction.</p>
<p>Deductive reasoning can be further subdivided into propositional deduction (reasoning about the truth values of propositions) and predicate-logic deduction (reasoning about object properties and relations). In LVLM contexts, both forms are relevant, with predicate-level deduction being particularly important for reasoning about visual scenes that involve multiple objects and their interactions [<xref ref-type="bibr" rid="ref-74">74</xref>].</p>
<p><bold>Inductive Reasoning</bold>. Inductive reasoning involves discovering general patterns or rules from specific instances or observations [<xref ref-type="bibr" rid="ref-11">11</xref>]. This form of reasoning is fundamental to scientific discovery and learning from experience. In LVLM contexts, inductive reasoning tasks include completing visual patterns, generalizing from examples, and identifying underlying rules.</p>
<p>The RAVEN benchmark [<xref ref-type="bibr" rid="ref-22">22</xref>] represents a significant contribution to evaluating inductive reasoning in visual contexts, presenting matrices in which models must identify underlying transformation rules. RAVEN includes visual matrices with varied structures, requiring models to understand relations between elements and apply transformation rules to complete the pattern.</p>
<p>Inductive reasoning in visual contexts presents unique challenges due to the complexity of visual representations and the need to handle variations in style, viewpoint, and context [<xref ref-type="bibr" rid="ref-46">46</xref>]. Unlike formal inductive reasoning in logic, visual induction often requires dealing with noisy, high-dimensional input and may involve probabilistic rather than deterministic generalization.</p>
<p>Inductive reasoning can be characterized by several key properties: the resulting generalization is uncertain (the conclusion may be false even if the observations are correct), the reasoning moves from specific observations to broader rules, and the strength of the generalization depends on the representativeness of the observed instances [<xref ref-type="bibr" rid="ref-11">11</xref>]. These properties make inductive reasoning fundamentally different from deductive reasoning and require different computational approaches.</p>
<p><bold>Abductive Reasoning</bold>. Abductive reasoning, also known as inference to the best explanation, involves inferring the most plausible explanation for observed phenomena [<xref ref-type="bibr" rid="ref-9">9</xref>,<xref ref-type="bibr" rid="ref-23">23</xref>,<xref ref-type="bibr" rid="ref-75">75</xref>]. In LVLM contexts, abductive reasoning is particularly relevant for tasks such as visual explanation generation, where models must infer what events or conditions could have produced an observed scene.</p>
<p>Recent works such as KN-VLM [<xref ref-type="bibr" rid="ref-76">76</xref>] have focused on visual abductive reasoning, developing models that can infer hypotheses to explain visual contexts. This capability is essential for applications such as accident analysis and safety monitoring, where systems must infer what events could have led to an observed situation.</p>
<p>Abductive reasoning is particularly challenging because it requires both understanding observed phenomena and generating plausible explanations that account for them [<xref ref-type="bibr" rid="ref-23">23</xref>]. Unlike deductive reasoning, where conclusions follow necessarily from premises, abductive reasoning involves selecting the most plausible explanation among multiple candidates, often under incomplete or uncertain information.</p>
<p><bold>Multi-Hop Reasoning</bold>. Multi-hop reasoning refers to the structural requirement that evidence or sub-goals be chained across multiple steps [<xref ref-type="bibr" rid="ref-54">54</xref>,<xref ref-type="bibr" rid="ref-56">56</xref>,<xref ref-type="bibr" rid="ref-57">57</xref>]. In this survey, it is treated as a cross-cutting reasoning pattern rather than a purely separate operator on par with deduction or abduction. A task is multi-hop when the model must maintain intermediate states and compose dispersed evidence before the final answer can be derived.</p>
<p>The CLEVR benchmark [<xref ref-type="bibr" rid="ref-39">39</xref>] represents a foundational contribution to multi-hop reasoning evaluation in visual question answering. CLEVR presents synthetic scenes with multiple objects and asks questions that require multi-step reasoning to answer. For example, a question might ask, &#x201C;What is the color of the object that is to the left of the red sphere?&#x201D;, which requires first identifying the red sphere, then finding objects to its left, and finally extracting the relevant color attribute.</p>
<p>This distinction is important for separating reasoning from simple retrieval. If a system merely fetches several relevant snippets or regions and one of them directly contains the answer, the process is multi-source retrieval but not necessarily multi-hop reasoning. By contrast, reasoning is required when retrieved pieces must be compared, composed, or constrained to infer a conclusion that is not explicitly stated in any single source. Multi-hop reasoning is therefore particularly relevant for real-world applications in which questions require integrating information across different parts of an image or across multiple images [<xref ref-type="bibr" rid="ref-36">36</xref>]. The ability to maintain intermediate conclusions and chain reasoning steps is essential for complex question answering, narrative understanding, and comprehensive scene analysis.</p>
<p><bold>Causal Reasoning</bold>. Causal reasoning involves identifying causal relationships between events or variables, distinguishing correlation from causation, and predicting the effects of interventions [<xref ref-type="bibr" rid="ref-10">10</xref>,<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>]. This form of reasoning is essential for understanding how changes in one variable affect others and for predicting the outcomes of actions.</p>
<p>Recent works such as the Multimodal Causal Reasoning Benchmark [<xref ref-type="bibr" rid="ref-10">10</xref>] have developed benchmarks specifically targeting causal reasoning in multimodal contexts. These benchmarks challenge LVLMs to identify causal links across modalities, requiring the joint understanding of visual and textual information.</p>
<p>Causal reasoning is fundamental to many real-world applications, such as autonomous driving (predicting the effects of actions) [<xref ref-type="bibr" rid="ref-14">14</xref>]. The ability to distinguish causal relationships from simple correlations is a key capability that separates robust reasoning systems from superficial pattern-matching systems.</p>
<p>Causal reasoning in LVLMs faces unique challenges due to the need to handle both visual and textual information about causal relationships [<xref ref-type="bibr" rid="ref-10">10</xref>,<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>]. Visual information may show how events unfold over time, while textual information may describe causal mechanisms or provide explanatory context. Integrating these modalities to perform causal reasoning requires sophisticated representation learning and inference capabilities [<xref ref-type="bibr" rid="ref-10">10</xref>].</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Conceptual Boundaries of Complex Logical Reasoning</title>
<p>To avoid ambiguity in the subsequent taxonomy, we first clarify the conceptual boundaries between complex logical reasoning and related paradigms. Despite the increasing use of the term <italic>complex logical reasoning</italic> in LVLM research, its conceptual boundaries remain ambiguous and often overlap with related notions such as commonsense reasoning, mathematical reasoning, and visual perception [<xref ref-type="bibr" rid="ref-26">26</xref>,<xref ref-type="bibr" rid="ref-33">33</xref>,<xref ref-type="bibr" rid="ref-77">77</xref>]. This ambiguity may lead to inconsistent evaluation and weaken the validity of taxonomic frameworks. To address this issue, we provide a structured clarification of how complex logical reasoning differs from, and interacts with, these closely related paradigms.</p>
<p>We define <italic>complex logical reasoning</italic> as the ability to perform multi-step, structured inference over multimodal inputs, where intermediate reasoning steps are required to derive the final answer [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>]. More specifically, a task falls into this scope when it requires at least one inferential transformation beyond direct lookup, such as rule application, evidence composition, contradiction resolution, or hypothesis comparison. This distinguishes it from <italic>perception</italic>, which primarily involves recognizing or extracting information from visual inputs (e.g., object detection or attribute identification), and from <italic>commonsense reasoning</italic>, which relies on implicit world knowledge and prior experience rather than explicit multi-step deduction [<xref ref-type="bibr" rid="ref-33">33</xref>]. It also clarifies the relation to <italic>compositional reasoning</italic>: compositional structure is often a useful ingredient, but it becomes complex logical reasoning only when composed parts must support a new inferential conclusion rather than a direct structural readout. In contrast, <italic>mathematical reasoning</italic> represents a more formalized and symbolic subset of logical reasoning, often involving precise calculations, algebraic manipulation, or rule-based derivations [<xref ref-type="bibr" rid="ref-15">15</xref>,<xref ref-type="bibr" rid="ref-26">26</xref>]. While mathematical reasoning is inherently logical, it typically operates in a more constrained and well-defined problem space compared to general multimodal reasoning [<xref ref-type="bibr" rid="ref-78">78</xref>,<xref ref-type="bibr" rid="ref-79">79</xref>].</p>
<p>Importantly, these paradigms are not mutually exclusive but instead exhibit hierarchical and compositional relationships. Complex logical reasoning often builds upon perceptual grounding and commonsense knowledge, while also incorporating mathematical reasoning in specialized tasks [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-13">13</xref>]. For example, solving a visual math problem may require perception (reading a diagram), commonsense (interpreting context), and symbolic reasoning (performing calculations) [<xref ref-type="bibr" rid="ref-80">80</xref>,<xref ref-type="bibr" rid="ref-81">81</xref>]. Therefore, rather than treating these categories as strictly separate, we view complex logical reasoning as an integrative capability that coordinates multiple cognitive components [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-77">77</xref>]. <xref ref-type="fig" rid="fig-4">Fig. 4</xref> further illustrates these conceptual boundaries by contrasting their input focus, core operation, evidence type, and typical failure mode.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Conceptual boundaries among perception, commonsense reasoning, mathematical reasoning, and complex logical reasoning in LVLM research. Complex logical reasoning is distinguished by its dependence on multi-step inference over multimodal premises, although it may incorporate perceptual grounding, commonsense priors, and mathematical operations as supporting components.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-4.tif"/>
</fig>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Relationship with Other Reasoning Paradigms</title>
<p>As shown in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>, complex logical reasoning in LVLMs is closely related to several other reasoning paradigms, including commonsense, temporal, spatial, counterfactual, and strategic reasoning [<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>]. Although these paradigms emphasize different aspects of inference, they frequently interact in practical multimodal tasks. Complex logical reasoning provides a structured inference framework, while the other paradigms contribute complementary capabilities such as background knowledge, temporal ordering, spatial constraint modeling, hypothetical analysis, and decision-oriented planning. Understanding these relationships is important for developing comprehensive reasoning systems that can handle the full range of real-world applications.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Relationship between logical reasoning and other reasoning paradigms in LVLMs. Logical reasoning serves as a core capability for multimodal inference, but it is closely connected with several related paradigms, including commonsense, temporal, spatial, counterfactual, and strategic reasoning.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-5.tif"/>
</fig>
<p><bold>Commonsense Reasoning</bold>. Commonsense reasoning involves applying everyday knowledge about the world to interpret situations and make reasonable inferences [<xref ref-type="bibr" rid="ref-33">33</xref>]. It frequently overlaps with logical reasoning, particularly abductive reasoning [<xref ref-type="bibr" rid="ref-9">9</xref>,<xref ref-type="bibr" rid="ref-23">23</xref>], when explaining observed phenomena. For example, if we see a wet sidewalk and dark clouds in the sky, commonsense reasoning allows us to infer that it likely rained.</p>
<p>Commonsense reasoning differs from formal logical reasoning in that it relies on implicit knowledge about the world rather than explicit logical rules [<xref ref-type="bibr" rid="ref-25">25</xref>]. This implicit knowledge is difficult to formalize and may vary across cultures and contexts. In LVLM contexts, commonsense reasoning is essential for interpreting ambiguous visual scenes and making reasonable assumptions about unobserved elements [<xref ref-type="bibr" rid="ref-82">82</xref>].</p>
<p>The relationship between commonsense reasoning and logical reasoning is complementary: logical reasoning provides formal frameworks for inference, while commonsense reasoning provides the background knowledge needed to apply those frameworks appropriately [<xref ref-type="bibr" rid="ref-24">24</xref>]. Integrating these reasoning types remains a key challenge for LVLM development.</p>
<p><bold>Temporal Reasoning</bold>. Temporal reasoning focuses on understanding time-ordered sequences of events and reasoning about time-dependent relationships [<xref ref-type="bibr" rid="ref-83">83</xref>,<xref ref-type="bibr" rid="ref-84">84</xref>]. In multimodal contexts, temporal reasoning is essential for understanding video content, narrative descriptions, and cause-effect relationships that unfold over time.</p>
<p>Temporal reasoning in LVLMs requires representing time and reasoning about temporal relations such as before, after, simultaneous, and during [<xref ref-type="bibr" rid="ref-85">85</xref>]. This capability is essential for tasks such as video question answering, where questions may concern the order of events or the duration of activities. Beyond explicitly reasoning-oriented video benchmarks, large-scale video understanding resources such as YouTube-8M [<xref ref-type="bibr" rid="ref-86">86</xref>] have also played an important enabling role by providing millions of annotated videos for scalable representation learning.</p>
<p>The integration of temporal reasoning with logical reasoning is particularly important for causal reasoning tasks, which require understanding how events unfold over time and identifying causal relationships between temporally ordered events [<xref ref-type="bibr" rid="ref-10">10</xref>,<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>].</p>
<p><bold>Spatial Reasoning</bold>. Spatial reasoning deals with understanding and reasoning about physical space and spatial relationships between objects [<xref ref-type="bibr" rid="ref-87">87</xref>]. In multimodal contexts, spatial reasoning is often combined with logical deduction to answer questions about object relationships, spatial configurations, and layout understanding.</p>
<p>Spatial reasoning in LVLMs requires representing spatial relationships such as above, below, inside, outside, near, and far [<xref ref-type="bibr" rid="ref-88">88</xref>]. Chen et al. [<xref ref-type="bibr" rid="ref-89">89</xref>] highlights that multimodal reasoning also requires explicit referential grounding: by allowing spatial coordinates to be expressed directly in natural-language interaction, it extends MLLMs toward location-aware dialogue and grounded reasoning. GeoQA is a representative benchmark in this direction, as it explicitly evaluates geometric question answering that requires joint visual perception, diagram understanding, and numerical reasoning [<xref ref-type="bibr" rid="ref-81">81</xref>].</p>
<p>The integration of spatial reasoning with logical reasoning is particularly important for tasks that require reasoning about spatial constraints, such as path planning, object arrangement, and spatial inference tasks that combine visual evidence with textual premises about spatial relationships [<xref ref-type="bibr" rid="ref-58">58</xref>].</p>
<p><bold>Counterfactual Reasoning</bold>. Counterfactual reasoning involves reasoning about hypothetical alternatives to past events. It requires considering what would have happened if circumstances had been different, a capability that is essential for planning, decision-making, and understanding causality [<xref ref-type="bibr" rid="ref-90">90</xref>].</p>
<p>Counterfactual reasoning is closely related to causal reasoning, as both involve understanding causal relationships [<xref ref-type="bibr" rid="ref-10">10</xref>,<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>]. However, counterfactual reasoning specifically focuses on hypothetical scenarios that did not actually occur, whereas causal reasoning may also concern actual causal relationships.</p>
<p>In LVLM contexts, counterfactual reasoning is important for tasks such as visual explanation generation, where models must consider alternative explanations for observed phenomena [<xref ref-type="bibr" rid="ref-23">23</xref>], and for decision support systems, where models must evaluate the outcomes of different actions.</p>
<p>These reasoning paradigms interact with logical reasoning in complex ways, and successful LVLM systems must integrate capabilities across multiple paradigms [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-24">24</xref>]. Realistic environments such as WebArena [<xref ref-type="bibr" rid="ref-91">91</xref>] highlight the importance of long-horizon reasoning, interaction planning, and functional task completion in complex environments. The complexity of real-world reasoning tasks often requires combining multiple reasoning types, making the development of comprehensive reasoning systems a significant challenge.</p>
<p>Another related paradigm is strategic reasoning, which concerns planning and decision-making under dynamic, uncertain, and often multi-agent environments. A recent survey [<xref ref-type="bibr" rid="ref-20">20</xref>] shows that strategic reasoning emphasizes anticipating others&#x2019; behavior and adaptive policy selection, making it related to the logical-inference focus of many LVLM studies.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>LVLM Architectures for Logical Reasoning</title>
<p>The architecture of a Large Vision-Language Model fundamentally shapes its capability to perform logical reasoning [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-6">6</xref>,<xref ref-type="bibr" rid="ref-40">40</xref>,<xref ref-type="bibr" rid="ref-92">92</xref>]. The choice of architecture involves trade-offs between expressiveness, training efficiency, interpretability, and the ability to handle complex reasoning tasks.</p>
<p>The development of LVLM architectures for logical reasoning has been driven by several key considerations. First, logical reasoning often requires maintaining intermediate conclusions and chaining reasoning steps, which poses challenges for standard sequence-to-sequence architectures [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-57">57</xref>]. Second, logical reasoning may benefit from explicit symbolic representations that can be verified and manipulated, which may not emerge naturally from neural architectures [<xref ref-type="bibr" rid="ref-41">41</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>]. Third, the multimodal nature of LVLMs introduces challenges in aligning visual and textual representations to support reasoning across modalities [<xref ref-type="bibr" rid="ref-59">59</xref>&#x2013;<xref ref-type="bibr" rid="ref-61">61</xref>].</p>
<p><xref ref-type="fig" rid="fig-6">Fig. 6</xref> provides a unified framework view of LVLM architectures for logical reasoning. At a high level, reasoning capability is shaped by two complementary aspects: (1) the architectural paradigm that organizes perception, language understanding, and reasoning, i.e., <italic>unified</italic>, <italic>modular</italic>, <italic>tool-augmented</italic>, and (2) the visual feature encoding and alignment mechanisms that determine how visual evidence is represented and grounded across modalities i.e., <italic>visual tokenization</italic>, <italic>cross-modal alignment</italic>, and <italic>spatial encoding</italic>.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Framework view of LVLM architectures for logical reasoning. Upper layer: three architectural paradigms (unified, modular, tool-augmented). Lower layer: visual foundations (tokenization, cross-modal alignment, spatial encoding). Together, they shape LVLM reasoning under complex multimodal inputs.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-6.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Architectural Design Paradigms</title>
<p>We identify three primary architectural paradigms for logical reasoning in LVLMs: unified architectures, modular architectures, and tool-augmented architectures. Although the boundaries between these paradigms are not always strict, they reflect three distinct design philosophies for integrating visual perception, language understanding, and reasoning. Unified architectures emphasize end-to-end multimodal representation learning; modular architectures explicitly separate perception and reasoning; and tool-augmented architectures extend LVLMs with external computation or symbolic tools. This categorization provides a useful lens for comparing how different systems support logical inference under multimodal inputs. <xref ref-type="table" rid="table-2">Table 2</xref> summarizes representative architecture works under these paradigms and highlights their reasoning-oriented design choices and limitations.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Representative LVLM architecture works for reasoning (2022&#x2013;early 2026).</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>Model</th>
<th>Year</th>
<th>Architecture</th>
<th>Reasoning-Oriented Design</th>
<th>Limitation</th>
</tr>
</thead>
<tbody>
<tr>
<td>Flamingo [<xref ref-type="bibr" rid="ref-93">93</xref>]</td>
<td>2022</td>
<td>Unified multimodal model with cross-attention</td>
<td>Frozen backbone with perceiver resampler enables few-shot multimodal in-context reasoning</td>
<td>Reasoning processes remain largely implicit</td>
</tr>
<tr>
<td>BLIP-2 [<xref ref-type="bibr" rid="ref-40">40</xref>,<xref ref-type="bibr" rid="ref-94">94</xref>]</td>
<td>2023</td>
<td>Unified bridge architecture (Q-Former)</td>
<td>Query bottleneck improves efficient vision-language alignment for downstream reasoning</td>
<td>Limited explicit reasoning verification</td>
</tr>
<tr>
<td>LLaVA [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-53">53</xref>]</td>
<td>2023</td>
<td>Instruction-tuned LVLM</td>
<td>Instruction data enhances compositional reasoning and step decomposition</td>
<td>Susceptible to hallucination under complex reasoning tasks</td>
</tr>
<tr>
<td>InternVL [<xref ref-type="bibr" rid="ref-95">95</xref>]</td>
<td>2023</td>
<td>Scaled vision backbone with multimodal alignment</td>
<td>Strong visual scaling improves OCR, grounding, and long-context perception</td>
<td>High computational cost</td>
</tr>
<tr>
<td>MiniGPT-4 [<xref ref-type="bibr" rid="ref-96">96</xref>]</td>
<td>2023</td>
<td>Frozen vision encoder &#x002B; projection &#x002B; LLM</td>
<td>Lightweight alignment unlocks strong multimodal understanding abilities</td>
<td>Outputs may be fluent but not always reliably grounded</td>
</tr>
<tr>
<td>Qwen-VL [<xref ref-type="bibr" rid="ref-7">7</xref>]</td>
<td>2023</td>
<td>Unified LVLM with visual receptor and grounding-aware interface</td>
<td>Strengthens text reading, grounding, and multilingual multimodal reasoning</td>
<td>Multi-step reasoning is not the primary optimization target</td>
</tr>
<tr>
<td>Otter [<xref ref-type="bibr" rid="ref-97">97</xref>]</td>
<td>2023</td>
<td>Multimodal in-context LLM</td>
<td>Supports interleaved image-text instruction following and in-context reasoning</td>
<td>Limited robustness in complex reasoning tasks</td>
</tr>
<tr>
<td>Kosmos-2 [<xref ref-type="bibr" rid="ref-98">98</xref>]</td>
<td>2023</td>
<td>Grounded multimodal LLM with location-token linking</td>
<td>Integrates grounding directly into generation, supporting referring and spatially grounded reasoning</td>
<td>Grounded generation increases interface complexity and does not itself ensure deep logical reasoning</td>
</tr>
<tr>
<td>Qwen2-VL [<xref ref-type="bibr" rid="ref-99">99</xref>]</td>
<td>2024</td>
<td>Unified multimodal LLM with Naive Dynamic Resolution and M-RoPE</td>
<td>Dynamic-resolution visual tokenization and unified image/video modeling improve document, chart, and video reasoning</td>
<td>Primarily optimized for benchmark performance rather than interpretable reasoning processes</td>
</tr>
<tr>
<td>LLaVA-OneVision [<xref ref-type="bibr" rid="ref-100">100</xref>]</td>
<td>2024</td>
<td>Unified multi-scenario model</td>
<td>Joint training across image and video reasoning settings</td>
<td>Limited robustness under distribution shifts</td>
</tr>
<tr>
<td>Chameleon [<xref ref-type="bibr" rid="ref-71">71</xref>]</td>
<td>2023</td>
<td>Plug-and-play LLM with external tools</td>
<td>Compositional reasoning with tool integration (web, vision, Python)</td>
<td>High complexity due to tool dependencies and planning</td>
</tr>
<tr>
<td>Reasoner/Reasoner-v2 [<xref ref-type="bibr" rid="ref-23">23</xref>]</td>
<td>2024</td>
<td>Cascaded reasoning transformer</td>
<td>Multi-level abductive reasoning for visual tasks</td>
<td>Limited applicability beyond visual abductive settings</td>
</tr>
<tr>
<td>Molmo [<xref ref-type="bibr" rid="ref-69">69</xref>]</td>
<td>2024</td>
<td>Open VLM trained with open data and curated pipeline</td>
<td>High-quality open pretraining/fine-tuning data improves robust multimodal reasoning without proprietary teacher models</td>
<td>Performance still depends strongly on data curation and scaling strategy</td>
</tr>
<tr>
<td>InternVL2.5 [<xref ref-type="bibr" rid="ref-101">101</xref>]</td>
<td>2024</td>
<td>Scaled InternVL family with model/data/test-time scaling</td>
<td>Explicitly benefits from CoT and test-time scaling, improving multidisciplinary and long-context reasoning</td>
<td>Larger-scale inference and test-time reasoning increase cost</td>
</tr>
<tr>
<td>LLaVA-NeXT-Interleave [<xref ref-type="bibr" rid="ref-72">72</xref>]</td>
<td>2024</td>
<td>Interleaved multimodal architecture for multi-image, video, and 3D</td>
<td>Treats interleaved multimodal context as a general template, improving cross-instance reasoning</td>
<td>More complex training recipe and evaluation setting</td>
</tr>
<tr>
<td>Qwen2.5-VL [<xref ref-type="bibr" rid="ref-73">73</xref>]</td>
<td>2025</td>
<td>Dynamic-resolution ViT &#x002B; unified multimodal LLM</td>
<td>Native dynamic-resolution perception, window attention, and long-video modeling improve document, chart, GUI, and agentic reasoning</td>
<td>Strong answer performance, but explicit reasoning faithfulness is not always guaranteed</td>
</tr>
<tr>
<td>Vision-R1 [<xref ref-type="bibr" rid="ref-70">70</xref>]</td>
<td>2025</td>
<td>Reasoning-oriented MLLM with cold-start CoT data &#x002B; RL</td>
<td>Uses multimodal CoT cold-start data, PTST, and GRPO to stimulate complex multimodal reasoning</td>
<td>Primarily optimized for reasoning benchmarks, especially math-heavy settings</td>
</tr>
<tr>
<td>R1-OneVision [<xref ref-type="bibr" rid="ref-65">65</xref>]</td>
<td>2025</td>
<td>Cross-modal formalization pipeline &#x002B; SFT/RL reasoning model</td>
<td>Transforms visual input into formal textual representations to enable step-by-step language-style reasoning</td>
<td>Formalization pipeline adds complexity and may lose raw visual nuance</td>
</tr>
<tr>
<td>WeThink [<xref ref-type="bibr" rid="ref-68">68</xref>]</td>
<td>2025</td>
<td>RL-enhanced vision-language reasoning framework</td>
<td>Builds a 120K reasoning-path dataset from 18 sources and applies hybrid-reward RL for general-purpose visual reasoning</td>
<td>Gains depend strongly on synthesized reasoning-data quality</td>
</tr>
<tr>
<td>GLM-4.1V-Think/GLM-4.5V [<xref ref-type="bibr" rid="ref-102">102</xref>]</td>
<td>2025</td>
<td>Scaled VLM family with reasoning-centric training &#x002B; RLCS</td>
<td>Curriculum-based RL unlocks broad reasoning ability across STEM, video, grounding, GUI agents, and long documents</td>
<td>Training and inference recipes are comparatively heavy and complex</td>
</tr>
<tr>
<td>MMaDA [<xref ref-type="bibr" rid="ref-103">103</xref>]</td>
<td>2025</td>
<td>Unified multimodal diffusion language model</td>
<td>Modality-agnostic diffusion architecture, mixed long CoT fine-tuning, and UniGRPO unify reasoning and generation</td>
<td>Less standard than autoregressive LVLMs and harder to compare directly with mainstream VLM pipelines</td>
</tr>
<tr>
<td><inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:msup><mml:mi>VLM-R</mml:mi><mml:mn>3</mml:mn></mml:msup></mml:math></inline-formula> [<xref ref-type="bibr" rid="ref-104">104</xref>]</td>
<td>2025</td>
<td>Region-recognition-and- reasoning framework with R-GRPO</td>
<td>Learns when and where to look, then injects cropped evidence back into interleaved CoT</td>
<td>Extra region-selection loop increases system complexity</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><bold>Unified Architectures</bold>. Unified architectures process visual and textual inputs within a tightly integrated multimodal model and learn joint representations across modalities. In practice, this family includes models that rely on cross-attention connectors, alignment bridges, or unified multimodal token spaces, even when the vision encoder and language backbone are not literally a single encoder. Representative examples include Flamingo [<xref ref-type="bibr" rid="ref-93">93</xref>], BLIP-2 [<xref ref-type="bibr" rid="ref-40">40</xref>], MiniGPT-4 [<xref ref-type="bibr" rid="ref-96">96</xref>], LLaVA [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-53">53</xref>], and more recent systems such as Qwen2-VL [<xref ref-type="bibr" rid="ref-99">99</xref>] and Qwen2.5-VL [<xref ref-type="bibr" rid="ref-73">73</xref>].</p>
<p>Flamingo [<xref ref-type="bibr" rid="ref-93">93</xref>] is an early representative of this paradigm. By inserting cross-attention layers between frozen visual and language backbones, it enables multimodal few-shot in-context learning while preserving the capabilities of large pretrained components. BLIP-2 [<xref ref-type="bibr" rid="ref-40">40</xref>] further improves this design through the Q-Former, which acts as a compact interface between visual features and the language model. LLaVA [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-53">53</xref>] shows that visual instruction tuning can substantially improve multimodal dialogue and compositional reasoning, while MiniGPT-4 [<xref ref-type="bibr" rid="ref-96">96</xref>] suggests that part of multimodal reasoning ability can be elicited from strong language backbones through lightweight alignment alone. More recent unified models also strengthen document, chart, multi-image, and video reasoning by improving visual tokenization, long-context handling, and interleaved multimodal modeling [<xref ref-type="bibr" rid="ref-72">72</xref>,<xref ref-type="bibr" rid="ref-73">73</xref>,<xref ref-type="bibr" rid="ref-99">99</xref>,<xref ref-type="bibr" rid="ref-100">100</xref>].</p>
<p>The main strength of unified architectures lies in end-to-end multimodal optimization. Because perception, alignment, and generation are trained in a closely coupled manner, these models are often effective at compositional understanding and at tasks that can be acquired through large-scale multimodal instruction tuning [<xref ref-type="bibr" rid="ref-2">2</xref>,<xref ref-type="bibr" rid="ref-16">16</xref>]. However, this tight coupling also makes the reasoning process difficult to interpret or verify. Intermediate reasoning is usually implicit in hidden states rather than represented as explicit symbolic structure, which can make these models less reliable on tasks requiring strict rule application, formal verification, or faithful multi-step inference [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-44">44</xref>].</p>
<p><bold>Modular Architectures</bold>. Modular architectures separate visual encoding, language understanding, and logical reasoning into distinct components. This separation allows each module to be optimized for its own role and makes it easier to incorporate specialized reasoning mechanisms when explicit inference is required.</p>
<p>A prominent instance of this paradigm is neuro-symbolic modeling, which combines neural perception with symbolic reasoning engines [<xref ref-type="bibr" rid="ref-41">41</xref>,<xref ref-type="bibr" rid="ref-63">63</xref>]. In such systems, visual and linguistic inputs are first parsed into structured intermediate representations, such as objects, relations, or logical forms. A symbolic module then performs explicit inference over these structures and produces conclusions that can be translated back into natural language. This approach is especially attractive for deductive reasoning and compositional visual reasoning because the reasoning procedure is more transparent and, in principle, verifiable [<xref ref-type="bibr" rid="ref-49">49</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>].</p>
<p>The main advantage of modular architectures is interpretability. Because perception and reasoning are partly decoupled, one can inspect the intermediate representations and reason about where errors occur. This is particularly valuable for logical tasks in which correctness depends on precise relational structure rather than fluent answer generation. The main weakness, however, lies in the interface between continuous perceptual representations and discrete symbolic structures. Parsing errors, information loss, or misalignment between neural outputs and symbolic inputs can propagate through the pipeline and degrade reasoning quality [<xref ref-type="bibr" rid="ref-62">62</xref>,<xref ref-type="bibr" rid="ref-71">71</xref>]. As a result, modular systems may be more rigorous on structured reasoning tasks, but less flexible when dealing with ambiguous, noisy, or open-ended multimodal inputs.</p>
<p><bold>Tool-Augmented Architectures</bold>. Tool-augmented architectures extend LVLMs with external tools such as code interpreters, symbolic solvers, search modules, or task-specific vision components [<xref ref-type="bibr" rid="ref-105">105</xref>,<xref ref-type="bibr" rid="ref-106">106</xref>]. Instead of requiring the core model to internalize all reasoning skills, this paradigm treats the LVLM as a coordinator that decides when to call tools, how to formulate the calls, and how to integrate the returned results.</p>
<p>This paradigm is particularly effective when reasoning requires forms of computation that neural models do not perform reliably on their own, such as exact calculation, formal logical checking, or structured program execution [<xref ref-type="bibr" rid="ref-42">42</xref>,<xref ref-type="bibr" rid="ref-43">43</xref>]. Representative examples include Visual Programming [<xref ref-type="bibr" rid="ref-107">107</xref>], which combines LLM-based planning with external vision modules and executable intermediate steps, and Chameleon [<xref ref-type="bibr" rid="ref-71">71</xref>], which integrates multiple external tools into compositional reasoning workflows.</p>
<p>The primary strength of tool-augmented architectures is flexibility. By outsourcing specialized operations to external modules, these systems can extend reasoning capabilities without redesigning the base LVLM. A deeper limitation is that tool-augmented systems shift the burden of reasoning from internal model representations to external orchestration. This introduces a new class of failure modes, including incorrect tool selection, invalid program generation, and misinterpretation of tool outputs [<xref ref-type="bibr" rid="ref-43">43</xref>,<xref ref-type="bibr" rid="ref-108">108</xref>]. In complex multi-step scenarios, such errors can accumulate across reasoning stages, making the overall system brittle despite the correctness of individual tools.</p>
<p><bold>Comparative Synthesis across Paradigms</bold>. The current empirical evidence remains uneven across unified, modular, and tool-augmented architectures. Broad shared benchmark suites such as MMBench,bib31, MathVista, MuirBench, and Video-MME are currently dominated by unified or unified-style LVLMs, which means they provide the clearest evidence for general benchmark transfer under common reporting conventions (see <xref ref-type="sec" rid="s5_3">Section 5.3</xref>). In our standardized evidence map in <xref ref-type="sec" rid="s5_3">Section 5.3</xref>, the systems with directly comparable results across these shared benchmarks are all unified or unified-style models, such as GPT-4V, Qwen2.5-VL, LLaVA-OneVision, and InternVL2.5 [<xref ref-type="bibr" rid="ref-37">37</xref>,<xref ref-type="bibr" rid="ref-73">73</xref>,<xref ref-type="bibr" rid="ref-100">100</xref>,<xref ref-type="bibr" rid="ref-101">101</xref>]. By contrast, modular and neuro-symbolic systems are more often evaluated on structured settings such as CLEVR, GeoQA, or targeted reasoning-faithfulness studies, where their explicit intermediate representations improve inspectability and rule-level verification but reduce direct comparability with open-ended LVLM benchmarks [<xref ref-type="bibr" rid="ref-39">39</xref>,<xref ref-type="bibr" rid="ref-49">49</xref>,<xref ref-type="bibr" rid="ref-81">81</xref>]. Tool-augmented systems report concrete gains when exact computation, symbolic execution, or external retrieval is essential, especially in program-execution or math-oriented settings, yet these gains are inseparable from the quality of orchestration and from the assumptions imposed by external tools [<xref ref-type="bibr" rid="ref-42">42</xref>,<xref ref-type="bibr" rid="ref-105">105</xref>,<xref ref-type="bibr" rid="ref-107">107</xref>]. Accordingly, current literature supports three cautious conclusions: unified architectures offer the strongest evidence of broad benchmark coverage, modular architectures offer the strongest evidence of explicit intermediate-state verification on structured tasks, and tool-augmented architectures offer the strongest evidence of exactness when external computation is reliable. None of these observations should be overgeneralized into a universal ranking, because the supporting evaluations still differ substantially in task distribution, annotation format, and end-to-end system boundary.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Visual Feature Encoding and Alignment</title>
<p>The quality of visual feature encoding directly affects logical reasoning capability. To support multimodal inference, visual representations must preserve information that is relevant not only for recognition, but also for abstraction, relation modeling, and cross-modal grounding [<xref ref-type="bibr" rid="ref-52">52</xref>,<xref ref-type="bibr" rid="ref-109">109</xref>]. This subsection discusses three aspects that are particularly important for reasoning-oriented LVLMs: visual tokenization, cross-modal alignment, and spatial encoding. <xref ref-type="fig" rid="fig-7">Fig. 7</xref> presents these encoding and alignment components as a reasoning-oriented representation pipeline.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Visual feature encoding and alignment for reasoning-oriented LVLMs. Visual tokenization determines the granularity of visual evidence; cross-modal alignment connects visual and textual representations; and spatial encoding preserves layout, pairwise relations, and scene structure needed for logical reasoning.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-7.tif"/>
</fig>
<p><bold>Visual Tokenization</bold>. Visual tokenization determines the granularity and structure of the visual information that is made available to the language model. Different tokenization strategies support different forms of reasoning.</p>
<p><italic>Patch-based tokenization</italic>, widely used in ViT-style architectures and models such as CLIP, divides an image into fixed-size patches that are treated as visual tokens [<xref ref-type="bibr" rid="ref-52">52</xref>,<xref ref-type="bibr" rid="ref-110">110</xref>]. This design is simple and scalable, and it works well for general-purpose representation learning. However, fixed patch grids do not naturally align with semantically meaningful entities, which can make object-centric and relation-centric reasoning more difficult.</p>
<p><italic>Region-based tokenization</italic> instead derives features from detected objects or salient regions, making the visual representation more compatible with object-level reasoning [<xref ref-type="bibr" rid="ref-111">111</xref>,<xref ref-type="bibr" rid="ref-112">112</xref>]. This can be advantageous for tasks involving counting, comparison, and spatial or relational inference. Its limitation is that it depends on upstream detection quality and may omit contextual information that is not well captured by explicit object proposals.</p>
<p>A related alternative is to rely on <italic>segmentation-aware representations</italic>, where pixel-level or region-level semantic labels provide dense structural cues [<xref ref-type="bibr" rid="ref-113">113</xref>]. Such representations can support fine-grained reasoning about attributes and scene layout, but the predefined semantic categories may still not align well with the more abstract or relational categories required for logical reasoning.</p>
<p><bold>Cross-Modal Alignment</bold>. Cross-modal alignment is the mechanism that connects visual and textual representations so that reasoning can proceed across modalities. In reasoning-oriented LVLMs, alignment must do more than capture coarse semantic similarity; it must support the binding of entities, attributes, relations, and constraints between modalities.</p>
<p>One common strategy is <italic>contrastive alignment</italic>, as used in CLIP-style pretraining, where paired image-text examples are pulled together in representation space and unpaired examples are pushed apart [<xref ref-type="bibr" rid="ref-52">52</xref>]. This produces strong global semantic representations, but it does not necessarily capture the fine-grained alignments needed for logical inference.</p>
<p>A second strategy is <italic>cross-attention-based alignment</italic>, exemplified by Flamingo [<xref ref-type="bibr" rid="ref-93">93</xref>], where the language model dynamically attends to visual features during inference. This design is more flexible and can adapt the alignment process to the current question or context, which is beneficial for reasoning tasks that require selective use of visual evidence.</p>
<p>A third strategy is <italic>fusion-based integration</italic>, in which visual and textual representations are combined through learned transformation modules or joint multimodal spaces [<xref ref-type="bibr" rid="ref-114">114</xref>]. Fusion can enable richer multimodal interaction, but if it is too coarse, it may blur modality-specific structure that remains important for reasoning.</p>
<p><bold>Spatial Encoding</bold>. Spatial information is often central to logical reasoning, particularly for tasks involving object relations, geometric structure, physical layout, and scene configuration [<xref ref-type="bibr" rid="ref-88">88</xref>,<xref ref-type="bibr" rid="ref-115">115</xref>]. As a result, the way spatial structure is encoded has a direct impact on reasoning performance.</p>
<p><italic>Absolute position encoding</italic> represents where tokens occur in the visual field. This can help preserve layout information, but absolute coordinates alone are often insufficient for reasoning because they do not directly capture relations between entities.</p>
<p><italic>Relative position encoding</italic> instead models pairwise spatial relations such as left of, right of, above, and below. These relational encodings are generally more robust across changes in scale, viewpoint, and composition, and they are often more relevant for logical reasoning tasks.</p>
<p><italic>Spatial structure encoding</italic> aims to represent higher-order layout or scene organization, enabling reasoning about configurations rather than isolated pairwise relations [<xref ref-type="bibr" rid="ref-74">74</xref>]. This is especially important for tasks involving multi-object spatial constraints, embodied interaction, or scene-level planning.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Reasoning Mechanisms and Strategies</title>
<p>The development of reasoning mechanisms in LVLMs has been strongly influenced by progress in LLM reasoning, particularly chain-of-thought prompting and its extensions, while also adapting to the additional challenges introduced by multimodal evidence [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-57">57</xref>]. In contrast to purely textual settings, multimodal reasoning requires not only generating intermediate inference steps, but also grounding those steps in visual inputs and maintaining consistency between visual observations and textual conclusions.</p>
<p>The following subsections examine four broad categories of reasoning mechanisms: chain-of-thought methods, program- and neuro-symbolic approaches, self-correction and reflection strategies, and interpretability-oriented analysis of reasoning processes. These mechanisms are not mutually exclusive; in practice, many recent systems combine several of them. <xref ref-type="fig" rid="fig-8">Fig. 8</xref> maps these mechanism families onto the overall multimodal reasoning pipeline.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Overview of reasoning mechanisms and strategies in LVLMs for complex multimodal reasoning. The pipeline transforms multimodal inputs to answers via perception, intermediate inference, and generation. Four mechanisms intervene at different stages: chain-of-thought for stepwise decomposition, program-based methods for structured reasoning with external tools, self-correction for iterative refinement, and interpretability analysis for assessing faithfulness and robustness.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-8.tif"/>
</fig>
<sec id="s4_1">
<label>4.1</label>
<title>Chain-of-Thought and Its Extensions</title>
<p>Chain-of-thought (CoT) reasoning encourages models to generate explicit intermediate reasoning steps before producing a final answer [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-116">116</xref>]. Early studies demonstrate that providing intermediate reasoning steps can significantly improve performance across arithmetic, commonsense, and symbolic reasoning tasks [<xref ref-type="bibr" rid="ref-117">117</xref>,<xref ref-type="bibr" rid="ref-118">118</xref>]. In LVLMs, this idea has become one of the most influential mechanisms for improving multi-step inference, because it makes the reasoning trajectory more explicit and can partially reduce direct answer guessing based on superficial multimodal correlations.</p>
<p>For LVLMs, visual or multimodal CoT extends standard CoT by explicitly integrating visual observations into the reasoning chain [<xref ref-type="bibr" rid="ref-116">116</xref>,<xref ref-type="bibr" rid="ref-119">119</xref>]. A common pattern is to first identify salient visual evidence, then translate it into textual or semi-structured intermediate descriptions, and finally use these grounded observations to support subsequent inference. In this sense, multimodal CoT typically involves two coupled components.</p>
<p><bold>Visual Grounding in Reasoning</bold>. The first component is the extraction or verbalization of visual evidence that is relevant to the reasoning process. This may take the form of textual descriptions of objects, attributes, spatial relations, charts, or diagram elements that can then be incorporated into the reasoning chain [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-94">94</xref>]. The quality of this grounding step is critical: if the visual evidence is misidentified or inaccurately verbalized, subsequent reasoning may remain coherent in form while being unsupported by the input.</p>
<p><bold>Multimodal Reasoning Chains</bold>. The second component is the explicit incorporation of grounded visual evidence into intermediate inference steps [<xref ref-type="bibr" rid="ref-36">36</xref>,<xref ref-type="bibr" rid="ref-116">116</xref>,<xref ref-type="bibr" rid="ref-120">120</xref>]. Rather than simply appending an image caption before answering, effective multimodal CoT requires the model to connect observations to relational, compositional, or causal inferences. For example, a valid chain may first identify that one object lies to the left of another and then use this grounded relation to answer a spatial query.</p>
<p>Representative methods such as Visual-CoT [<xref ref-type="bibr" rid="ref-119">119</xref>] and Multimodal-CoT/MMCoT [<xref ref-type="bibr" rid="ref-116">116</xref>,<xref ref-type="bibr" rid="ref-120">120</xref>] show that explicitly structuring reasoning around grounded visual evidence can improve performance on complex multimodal reasoning tasks. A common design choice in these approaches is to separate rationale generation from answer prediction, so that the model first constructs a reasoning trajectory and only then produces the final response. This decomposition can improve transparency and sometimes accuracy, although it does not by itself guarantee reasoning faithfulness.</p>
<p>CoT has also been extended in more structured directions. For instance, compositional chain-of-thought prompting [<xref ref-type="bibr" rid="ref-36">36</xref>] incorporates scene-graph-style relational structure into intermediate reasoning, thereby encouraging the model to reason over object relations more explicitly. More broadly, multimodal CoT methods increasingly attempt to move from free-form natural language rationales toward more structured intermediate representations.</p>
<p>Despite its practical effectiveness, CoT has clear limitations. First, generated reasoning chains may be plausible but unfaithful, reflecting post-hoc rationalization rather than the true basis of the model&#x2019;s prediction [<xref ref-type="bibr" rid="ref-59">59</xref>,<xref ref-type="bibr" rid="ref-60">60</xref>]. Notably, prior empirical studies have shown that even invalid or partially incorrect reasoning chains can still yield high task performance, suggesting that CoT may not faithfully reflect the underlying reasoning process [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-117">117</xref>,<xref ref-type="bibr" rid="ref-118">118</xref>]. <xref ref-type="table" rid="table-3">Table 3</xref> summarizes representative reasoning mechanisms in LVLMs, highlighting their core features and the main limitations observed in current studies.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Representative reasoning mechanisms in LVLMs.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>Mechanism</th>
<th>Representative Works</th>
<th>Core Feature</th>
<th>Observed Limitation</th>
</tr>
</thead>
<tbody>
<tr>
<td>Multimodal CoT</td>
<td>Multimodal-CoT [<xref ref-type="bibr" rid="ref-116">116</xref>], Compositional Chain-of-Thought Prompting [<xref ref-type="bibr" rid="ref-36">36</xref>], Visual Chain of Thought [<xref ref-type="bibr" rid="ref-119">119</xref>], MM-CoT [<xref ref-type="bibr" rid="ref-120">120</xref>], Visual SKETCHPAD [<xref ref-type="bibr" rid="ref-58">58</xref>], Least-to-Most Prompting [<xref ref-type="bibr" rid="ref-121">121</xref>], Chain-of-Thought Prompting [<xref ref-type="bibr" rid="ref-1">1</xref>], Tree of Thoughts [<xref ref-type="bibr" rid="ref-57">57</xref>], EmbodiedVSR [<xref ref-type="bibr" rid="ref-74">74</xref>], VcCoT [<xref ref-type="bibr" rid="ref-10">10</xref>]</td>
<td>Explicitly externalizes intermediate reasoning steps with visual grounding</td>
<td>Verbose but potentially unfaithful reasoning</td>
</tr>
<tr>
<td>Program-of-Thought</td>
<td>PoT Prompting [<xref ref-type="bibr" rid="ref-43">43</xref>], ToRA [<xref ref-type="bibr" rid="ref-42">42</xref>], PAL [<xref ref-type="bibr" rid="ref-105">105</xref>], UniGeo [<xref ref-type="bibr" rid="ref-122">122</xref>], MathQA [<xref ref-type="bibr" rid="ref-123">123</xref>], Visual Programming [<xref ref-type="bibr" rid="ref-107">107</xref>], TheoremQA [<xref ref-type="bibr" rid="ref-28">28</xref>], GeoQA [<xref ref-type="bibr" rid="ref-81">81</xref>]</td>
<td>Converts reasoning into executable programs for verifiable intermediate states</td>
<td>Tool dependency and error propagation</td>
</tr>
<tr>
<td>Self-consistency/reflection</td>
<td>Self-Consistency [<xref ref-type="bibr" rid="ref-44">44</xref>], Masked Thought [<xref ref-type="bibr" rid="ref-124">124</xref>], Training Verifiers to Solve Math Word Problems [<xref ref-type="bibr" rid="ref-125">125</xref>], Math-Shepherd [<xref ref-type="bibr" rid="ref-126">126</xref>], WizardMath [<xref ref-type="bibr" rid="ref-17">17</xref>], DeepSeek-R1 [<xref ref-type="bibr" rid="ref-127">127</xref>], Reasoning-LM [<xref ref-type="bibr" rid="ref-38">38</xref>]</td>
<td>Aggregates multiple reasoning paths or iteratively refines intermediate outputs</td>
<td>Increased inference cost</td>
</tr>
<tr>
<td>Neuro-symbolic integration</td>
<td>LINC [<xref ref-type="bibr" rid="ref-62">62</xref>], NS-CL [<xref ref-type="bibr" rid="ref-63">63</xref>], NS-VQA [<xref ref-type="bibr" rid="ref-41">41</xref>], Neuro-VLMs [<xref ref-type="bibr" rid="ref-49">49</xref>], Neuro-Symbolic AI [<xref ref-type="bibr" rid="ref-51">51</xref>], Abstraction NNs [<xref ref-type="bibr" rid="ref-21">21</xref>]</td>
<td>Combines neural perception with symbolic constraints for structured reasoning</td>
<td>Symbol grounding challenges</td>
</tr>
<tr>
<td>Code-centric reasoning training</td>
<td>Code-Think [<xref ref-type="bibr" rid="ref-108">108</xref>], DeepSeek-R1 [<xref ref-type="bibr" rid="ref-127">127</xref>], Llemma [<xref ref-type="bibr" rid="ref-128">128</xref>], ToRA [<xref ref-type="bibr" rid="ref-42">42</xref>], PAL [<xref ref-type="bibr" rid="ref-105">105</xref>], PoT Prompting [<xref ref-type="bibr" rid="ref-43">43</xref>], OpenMathInstruct-2 [<xref ref-type="bibr" rid="ref-18">18</xref>], MAmmoTH2 [<xref ref-type="bibr" rid="ref-79">79</xref>]</td>
<td>Encourages structured intermediate computation via code-based reasoning</td>
<td>Limited generalization across domains</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Program-of-Thought and Neuro-Symbolic Approaches</title>
<p>Program-of-thought (PoT) approaches reformulate reasoning problems as executable programs, typically using code interpreters or symbolic runtimes to carry out intermediate computations [<xref ref-type="bibr" rid="ref-43">43</xref>,<xref ref-type="bibr" rid="ref-105">105</xref>]. This idea is closely related to interpretable reasoning datasets such as MathQA [<xref ref-type="bibr" rid="ref-123">123</xref>], in which solving a problem involves generating an explicit sequence of operations rather than only a final answer. In multimodal settings, PoT is especially appealing because it offers a way to move part of the reasoning process from opaque neural inference into verifiable procedural execution. <xref ref-type="fig" rid="fig-9">Fig. 9</xref> contrasts implicit neural reasoning with program-of-thought and neuro-symbolic reasoning pipelines.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>Program-of-thought and neuro-symbolic reasoning in LVLMs.Top: traditional neural reasoning relies on implicit hidden-state inference, which is difficult to verify. Middle: program-of-thought (PoT) converts reasoning into executable programs, enabling step-wise verification and precise computation. Bottom: neuro-symbolic approaches combine neural perception with symbolic inference, allowing explicit reasoning over structured representations.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-9.tif"/>
</fig>
<p>The key strength of PoT is that it externalizes reasoning into a structured form. Instead of relying solely on free-form natural language rationales, the model generates code or executable procedures whose intermediate states can be inspected and verified. This is particularly valuable for tasks involving arithmetic computation, symbolic manipulation, geometric reasoning, or other forms of structured inference in which correctness can be checked mechanically [<xref ref-type="bibr" rid="ref-43">43</xref>,<xref ref-type="bibr" rid="ref-105">105</xref>].</p>
<p>From a logical-reasoning perspective, PoT offers three major benefits. First, it makes reasoning <bold>explicit</bold>: each computational step is represented in a formal procedure rather than being hidden in model activations. Second, it adds <bold>computational precision</bold>: once the problem has been translated into an executable program, the system can rely on the external runtime for exact calculation or rule application. Third, it improves <bold>debuggability</bold>: errors can often be localized to specific steps in the generated program instead of being diffused across a natural-language rationale.</p>
<p>Recent work on code-enhanced reasoning [<xref ref-type="bibr" rid="ref-38">38</xref>,<xref ref-type="bibr" rid="ref-108">108</xref>] further suggests that executable intermediate representations can improve reasoning beyond purely mathematical settings. However, the success of PoT depends critically on whether the model can correctly formalize the underlying problem. In multimodal contexts, this often requires translating visual inputs into structured symbolic variables, relations, or functions before execution. If this translation fails, the advantages of downstream executability are greatly reduced.</p>
<p>Neuro-symbolic approaches address a related problem from a complementary direction. Rather than merely converting reasoning into code, they explicitly combine neural perception modules with symbolic inference engines [<xref ref-type="bibr" rid="ref-41">41</xref>,<xref ref-type="bibr" rid="ref-63">63</xref>]. In these systems, neural components handle perception and language understanding, while symbolic components operate over structured intermediate representations to perform formal inference. This architecture is particularly well suited to tasks such as deductive reasoning, compositional visual reasoning, and logical entailment, where the reasoning process benefits from discrete structure and explicit rule application [<xref ref-type="bibr" rid="ref-62">62</xref>].</p>
<p>A representative example is Visual Programming (VisProg) [<xref ref-type="bibr" rid="ref-107">107</xref>], which decomposes complex visual tasks into modular executable procedures. More broadly, recent reviews of neuro-symbolic AI [<xref ref-type="bibr" rid="ref-51">51</xref>] and studies of robust reasoning in VLMs [<xref ref-type="bibr" rid="ref-49">49</xref>] show that neuro-symbolic methods remain one of the most promising routes toward stronger reasoning faithfulness and verifiability.</p>
<p>At the same time, both PoT and neuro-symbolic methods face a common bottleneck: the interface between neural representations and symbolic structure. To benefit from formal reasoning, the model must first produce accurate structured representations from ambiguous and noisy multimodal inputs. Errors in parsing, grounding, or formalization can easily propagate into the symbolic stage. For this reason, these approaches are often strongest on structured reasoning tasks, but less robust in open-ended scenarios where perceptual ambiguity and linguistic underspecification are substantial.</p>
<p>This point is crucial when comparing these methods with CoT-style reasoning. Program execution and symbolic inference can improve <italic>step verifiability</italic> once a correct formal representation has been obtained, but they do not automatically guarantee <italic>reasoning faithfulness</italic> in the broader multimodal sense. In practice, PoT often verifies the correctness of downstream computation rather than the correctness of upstream visual formalization, while neuro-symbolic pipelines often verify rule application rather than the reliability of perceptual grounding. As a result, these methods reduce some forms of structured hallucination and arithmetic error, yet they do not fully eliminate the gap between answer correctness and genuinely grounded reasoning in open-ended LVLM settings [<xref ref-type="bibr" rid="ref-49">49</xref>,<xref ref-type="bibr" rid="ref-59">59</xref>,<xref ref-type="bibr" rid="ref-60">60</xref>]. Their main contribution is therefore best understood as relocating the faithfulness bottleneck from free-form generation to representation quality, grounding fidelity, and interface robustness.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Self-Correction and Reflection Mechanisms</title>
<p>Self-correction mechanisms aim to improve reasoning reliability by enabling models to detect, revise, or filter erroneous reasoning trajectories [<xref ref-type="bibr" rid="ref-44">44</xref>,<xref ref-type="bibr" rid="ref-57">57</xref>]. These mechanisms are particularly important for LVLMs because multimodal logical reasoning is vulnerable to two intertwined sources of error: perceptual mistakes in interpreting the input and inferential mistakes in chaining the intermediate steps. Without some form of internal correction or verification, such errors can propagate directly to the final answer.</p>
<p>One widely used strategy is <bold>self-consistency</bold> [<xref ref-type="bibr" rid="ref-44">44</xref>]. Instead of relying on a single reasoning chain, the model samples multiple reasoning paths and selects the answer that is most consistent across them. The intuition is that correct reasoning is more likely to converge to the same conclusion across diverse trajectories, whereas incorrect reasoning may produce unstable or conflicting outputs. In multimodal reasoning, this can reduce sensitivity to local reasoning errors or brittle prompt effects, although its effectiveness depends on the diversity and quality of the sampled chains.</p>
<p>A second strategy is <bold>reflection</bold> or <bold>reflexive self-evaluation</bold>, in which the model explicitly reviews its own reasoning before committing to a final answer [<xref ref-type="bibr" rid="ref-38">38</xref>]. Here, the model may be prompted to critique intermediate steps, reconsider missing evidence, or justify why the conclusion follows from the input. Reflection can be useful for identifying obvious inconsistencies or unsupported leaps, but it does not guarantee genuine error detection: a model may also rationalize an incorrect answer more confidently rather than correct it.</p>
<p>A third strategy is <bold>iterative refinement</bold>, where reasoning is updated across multiple rounds rather than produced in a single forward pass [<xref ref-type="bibr" rid="ref-57">57</xref>,<xref ref-type="bibr" rid="ref-121">121</xref>]. In these approaches, intermediate reasoning states are progressively revised, expanded, or reorganized. This is especially helpful for tasks requiring long reasoning chains or hierarchical decomposition, since early-stage drafts can be corrected before the final output is generated.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Interpretability of Reasoning Processes</title>
<p>Interpretability is crucial for understanding how LVLMs produce reasoning outputs and for diagnosing why they fail [<xref ref-type="bibr" rid="ref-59">59</xref>,<xref ref-type="bibr" rid="ref-60">60</xref>]. In the context of logical reasoning, interpretability is not merely a desirable auxiliary property; it is closely tied to the broader question of reasoning faithfulness. If a model produces a correct answer, we still need to understand whether the answer was derived through grounded inference, shortcut exploitation, or post-hoc explanation.</p>
<p>Several interpretability approaches are particularly relevant to multimodal logical reasoning.</p>
<p><bold>Attention Visualization</bold>. One common approach is to inspect attention patterns to determine which visual elements and textual tokens influence the model during reasoning [<xref ref-type="bibr" rid="ref-88">88</xref>]. In multimodal settings, attention analysis can help reveal whether the model is attending to the relevant objects, regions, or relations when producing a reasoning trace or final answer. Although attention alone should not be overinterpreted as a full explanation of model behavior, it remains a useful diagnostic signal for detecting missing grounding, distractor sensitivity, or cross-modal misalignment.</p>
<p><bold>Reasoning Step Tracking</bold>. Another important approach is to inspect intermediate reasoning states or generated step-by-step traces [<xref ref-type="bibr" rid="ref-120">120</xref>]. Tracking reasoning steps allows researchers to identify where a reasoning chain breaks down, whether the error originates from perception or inference, and how mistakes propagate across multiple steps. This is particularly valuable for tasks involving long-horizon reasoning, where final-answer evaluation alone provides little information about failure modes.</p>
<p><bold>Explanation Analysis</bold>. A third approach is to examine generated natural-language explanations or rationales [<xref ref-type="bibr" rid="ref-1">1</xref>]. Such explanations can provide insight into the model&#x2019;s apparent reasoning strategy and may reveal whether the conclusion is supported by grounded evidence or instead reflects superficial pattern matching. However, explanation generation also has a major limitation: plausible explanations can be post-hoc and unfaithful. As a result, explanations should be treated as evidence for analysis, not as definitive proof of the model&#x2019;s internal reasoning process.</p>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Benchmarks and Evaluation Protocols</title>
<p>Evaluating logical reasoning capabilities in LVLMs requires not only challenging benchmarks, but also evaluation protocols that can distinguish genuine reasoning from superficial answer matching. This section surveys representative benchmarks and evaluation practices, with a focus on how well current evaluation settings capture the diversity, depth, and faithfulness of complex logical reasoning [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>].</p>
<p>As shown in <xref ref-type="fig" rid="fig-10">Fig. 10</xref>, the evaluation of logical reasoning in LVLMs can be understood as a three-part framework. Existing benchmarks cover several recurring reasoning types, including inductive, compositional, deductive, and causal reasoning, but remain fragmented in coverage. On top of this benchmark landscape, evaluation protocols can be divided into answer-level and process-level metrics.</p>
<fig id="fig-10">
<label>Figure 10</label>
<caption>
<title>Framework view of benchmarks and evaluation protocols for logical reasoning in LVLMs, including both answer-level metrics and a structured set of process-level metrics (step correctness, faithfulness, consistency, completeness).</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-10.tif"/>
</fig>
<sec id="s5_1">
<label>5.1</label>
<title>Existing Benchmarks for Logical Reasoning</title>
<p>We categorize benchmarks according to the primary form of reasoning they evaluate. This organization is not absolute, since many benchmarks involve multiple reasoning skills simultaneously. Nonetheless, it provides a useful overview of the current evaluation landscape and helps clarify which aspects of logical reasoning are well covered and which remain underexplored. <xref ref-type="table" rid="table-4">Table 4</xref> provides a comparison of representative logical reasoning benchmarks, including their release year, primary reasoning type, modality, and core evaluation focus.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Comparison of logical reasoning benchmarks.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>Benchmark</th>
<th>Year</th>
<th>Reasoning Type</th>
<th>Modality</th>
<th>Core Focus</th>
</tr>
</thead>
<tbody>
<tr>
<td>VQA [<xref ref-type="bibr" rid="ref-8">8</xref>]</td>
<td>2015</td>
<td>General visual reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Open-ended visual question answering over natural images</td>
</tr>
<tr>
<td>CLEVR [<xref ref-type="bibr" rid="ref-39">39</xref>]</td>
<td>2016</td>
<td>Multi-hop</td>
<td>Visual</td>
<td>Compositional relational reasoning</td>
</tr>
<tr>
<td>DocVQA [<xref ref-type="bibr" rid="ref-129">129</xref>]</td>
<td>2021</td>
<td>Document reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Visual question answering over document images with layout and OCR dependency</td>
</tr>
<tr>
<td>GeoQA [<xref ref-type="bibr" rid="ref-81">81</xref>]</td>
<td>2022</td>
<td>Geometric reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Diagram-based geometric reasoning with numerical inference</td>
</tr>
<tr>
<td>ChartQA [<xref ref-type="bibr" rid="ref-130">130</xref>]</td>
<td>2022</td>
<td>Mathematical reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Chart understanding with visual and numerical reasoning</td>
</tr>
<tr>
<td>UniGeo [<xref ref-type="bibr" rid="ref-122">122</xref>]</td>
<td>2022</td>
<td>Geometric reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Unified benchmark for diagram-based geometric logical and numerical reasoning</td>
</tr>
<tr>
<td>SEED-Bench [<xref ref-type="bibr" rid="ref-30">30</xref>]</td>
<td>2023</td>
<td>Multi-type</td>
<td>Vision &#x002B; Language</td>
<td>Generative multimodal comprehension</td>
</tr>
<tr>
<td>MMBench [<xref ref-type="bibr" rid="ref-45">45</xref>]</td>
<td>2023</td>
<td>Multi-type</td>
<td>Vision &#x002B; Language</td>
<td>Broad multimodal capability evaluation</td>
</tr>
<tr>
<td>MMMU [<xref ref-type="bibr" rid="ref-31">31</xref>]</td>
<td>2023</td>
<td>Expert-level reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Discipline-level question answering</td>
</tr>
<tr>
<td>TheoremQA [<xref ref-type="bibr" rid="ref-28">28</xref>]</td>
<td>2023</td>
<td>Mathematical reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Theorem-driven scientific and mathematical reasoning with formal knowledge dependence</td>
</tr>
<tr>
<td>MathVista [<xref ref-type="bibr" rid="ref-78">78</xref>]</td>
<td>2024</td>
<td>Mathematical reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Diagram and chart-based reasoning</td>
</tr>
<tr>
<td>MMMU-Pro [<xref ref-type="bibr" rid="ref-31">31</xref>]</td>
<td>2024</td>
<td>Robust expert reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Advanced multimodal reasoning under increased difficulty</td>
</tr>
<tr>
<td>Video-MME [<xref ref-type="bibr" rid="ref-85">85</xref>,<xref ref-type="bibr" rid="ref-131">131</xref>]</td>
<td>2024</td>
<td>Temporal multi-hop</td>
<td>Video &#x002B; Text</td>
<td>Comprehensive video reasoning evaluation</td>
</tr>
<tr>
<td>MuirBench [<xref ref-type="bibr" rid="ref-132">132</xref>]</td>
<td>2024</td>
<td>Multi-image reasoning</td>
<td>Multi-image &#x002B; Text</td>
<td>Robust multi-image understanding</td>
</tr>
<tr>
<td>MathVerse [<xref ref-type="bibr" rid="ref-80">80</xref>]</td>
<td>2024</td>
<td>Mathematical reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Large-scale visual mathematical problem solving</td>
</tr>
<tr>
<td>OlympiadBench [<xref ref-type="bibr" rid="ref-27">27</xref>]</td>
<td>2024</td>
<td>Olympiad-level reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Challenging bilingual scientific problems requiring advanced multimodal reasoning</td>
</tr>
<tr>
<td>MME [<xref ref-type="bibr" rid="ref-29">29</xref>]</td>
<td>2024</td>
<td>Multi-type</td>
<td>Vision &#x002B; Language</td>
<td>Broad evaluation suite for multimodal perception and reasoning abilities</td>
</tr>
<tr>
<td>SEED-Bench-2 [<xref ref-type="bibr" rid="ref-30">30</xref>]</td>
<td>2024</td>
<td>Multi-type</td>
<td>Vision &#x002B; Language</td>
<td>Expanded multimodal benchmark with stronger coverage of reasoning and perception diversity</td>
</tr>
<tr>
<td>BRIDGE [<xref ref-type="bibr" rid="ref-55">55</xref>]</td>
<td>2024</td>
<td>Multi-hop</td>
<td>Document &#x002B; Vision &#x002B; Language</td>
<td>Multi-hop reasoning in long multimodal documents with grounded evidence</td>
</tr>
<tr>
<td>M3GQA [<xref ref-type="bibr" rid="ref-54">54</xref>]</td>
<td>2024</td>
<td>Multi-hop graph reasoning</td>
<td>Vision &#x002B; Graph &#x002B; Language</td>
<td>Multi-entity and multi-hop reasoning across graph-structured multimodal settings</td>
</tr>
<tr>
<td>MathWriting [<xref ref-type="bibr" rid="ref-133">133</xref>]</td>
<td>2025</td>
<td>Mathematical reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Handwritten mathematical expression understanding and reasoning</td>
</tr>
<tr>
<td>CausalVLBench [<xref ref-type="bibr" rid="ref-34">34</xref>]</td>
<td>2025</td>
<td>Causal reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Benchmarking causal attribution and intervention-oriented reasoning in LVLMs</td>
</tr>
<tr>
<td>MuCR [<xref ref-type="bibr" rid="ref-10">10</xref>]</td>
<td>2025</td>
<td>Causal reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Multimodal causal link identification across visual and textual evidence</td>
</tr>
<tr>
<td>VcCoT [<xref ref-type="bibr" rid="ref-10">10</xref>]</td>
<td>2025</td>
<td>Causal chain-of-thought reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Evaluating causal reasoning with explicit multimodal reasoning chains</td>
</tr>
<tr>
<td>InPhyRe [<xref ref-type="bibr" rid="ref-67">67</xref>]</td>
<td>2025</td>
<td>Inductive physical reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Evaluates whether multimodal models can infer physical rules and abstract patterns from examples</td>
</tr>
<tr>
<td>DixitWorld [<xref ref-type="bibr" rid="ref-75">75</xref>]</td>
<td>2025</td>
<td>Abductive reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Evaluates plausible explanation generation through multi-agent Dixit-style gameplay</td>
</tr>
<tr>
<td>Mind the Gap [<xref ref-type="bibr" rid="ref-87">87</xref>]</td>
<td>2025</td>
<td>Spatial reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Comprehensive benchmarking of spatial reasoning failures and gaps in vision-language models</td>
</tr>
<tr>
<td>MM-IQ [<xref ref-type="bibr" rid="ref-46">46</xref>]</td>
<td>2025</td>
<td>Abstraction and reasoning</td>
<td>Vision &#x002B; Language</td>
<td>Benchmarks human-like abstraction, analogy, and general reasoning in multimodal models</td>
</tr>
<tr>
<td>VCR-Bench [<xref ref-type="bibr" rid="ref-134">134</xref>]</td>
<td>2025</td>
<td>Video chain-of-thought reasoning</td>
<td>Video &#x002B; Text</td>
<td>Comprehensive evaluation of explicit reasoning chains in video understanding</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>At a broad level, current benchmarks fall into several recurring groups: <italic>inductive or analogy-style benchmarks</italic>, <italic>compositional and multi-hop benchmarks</italic>, <italic>deductive or entailment-oriented benchmarks</italic>, and <italic>causal or counterfactual benchmarks</italic>. In addition, a number of broader multimodal evaluation suites include logical reasoning as one component, though often not as their exclusive focus.</p>
<p><bold>Inductive reasoning benchmarks.</bold> RAVEN [<xref ref-type="bibr" rid="ref-22">22</xref>] is one of the most representative benchmarks for inductive reasoning in visual contexts. It evaluates rule induction through Raven-style matrix completion tasks, in which models must infer latent transformation rules from structured visual patterns. The benchmark includes multiple matrix configurations, such as relational and progressive structures, and therefore provides a controlled testbed for evaluating whether a model can abstract patterns rather than merely recognize objects. Its main strengths lie in its clear focus on rule induction and its systematic control of difficulty. Its main limitation, however, is that it is primarily visual and synthetic, which constrains its representativeness for broader multimodal logical reasoning.</p>
<p><bold>Compositional and multi-hop reasoning benchmarks.</bold> CLEVR [<xref ref-type="bibr" rid="ref-39">39</xref>] remains a foundational benchmark for compositional and multi-step visual reasoning. It presents synthetic scenes containing multiple objects and asks questions that require combining attributes, relations, and counting operations across several reasoning steps. CLEVR is valuable because it was explicitly designed to reduce dataset bias and isolate reasoning structure. However, its synthetic nature also limits ecological validity: the benchmark evaluates controlled compositional reasoning well, but does not capture the ambiguity and noise of real-world multimodal inputs.</p>
<p>Beyond synthetic scenes, more recent multi-hop settings attempt to evaluate more realistic evidence integration. Such benchmarks are motivated by the observation that many real reasoning tasks require models to connect information scattered across different regions, images, or modalities rather than answer from a single local cue [<xref ref-type="bibr" rid="ref-54">54</xref>&#x2013;<xref ref-type="bibr" rid="ref-56">56</xref>]. This broader class of benchmarks is particularly important for evaluating logical reasoning in document, chart, and multi-image settings.</p>
<p><bold>Deductive reasoning benchmarks.</bold> ReClor [<xref ref-type="bibr" rid="ref-64">64</xref>] is a representative benchmark for deductive reasoning and logical entailment. It presents reading comprehension problems that require identifying whether a conclusion follows from a set of premises through valid logical inference. Its main value lies in its emphasis on formal reasoning structure rather than open-ended commonsense interpretation. However, because it is fundamentally text-based, its relevance to multimodal LVLM evaluation is indirect: it is highly useful for conceptual grounding, but does not by itself test whether a model can integrate visual evidence into deductive inference.</p>
<p><bold>Causal reasoning benchmarks.</bold> Causal reasoning has only recently begun to receive dedicated multimodal benchmark support. MuCR (Multimodal Causal Reasoning Benchmark) [<xref ref-type="bibr" rid="ref-10">10</xref>] is a representative example. It evaluates whether models can identify causal relations across visual and textual evidence, distinguish correlation from causation, and reason about interventions or counterfactual changes. Compared with earlier general multimodal benchmarks, MuCR is notable because it explicitly targets a class of reasoning that is central to real-world decision making but difficult to capture with answer-only evaluation.</p>
<p>Related efforts extend causal evaluation to temporal settings. Video-based causal benchmarks assess whether models can infer causal relations from event sequences rather than static images alone [<xref ref-type="bibr" rid="ref-34">34</xref>]. Such tasks are especially important because causal reasoning often depends on temporal structure, yet they are also more difficult to evaluate reliably due to longer contexts and greater annotation complexity.</p>
<p><bold>Broader observations on current benchmark coverage.</bold> Taken together, current benchmarks cover important but still fragmented slices of the reasoning problem. RAVEN captures abstract pattern induction, CLEVR captures synthetic compositional reasoning, ReClor captures text-based deductive reasoning, and MuCR and related datasets begin to probe multimodal causal inference. However, few benchmarks simultaneously combine real-world multimodal complexity, diverse reasoning types, and process-aware evaluation. This fragmentation remains one of the central challenges in assessing logical reasoning in LVLMs [<xref ref-type="bibr" rid="ref-25">25</xref>,<xref ref-type="bibr" rid="ref-29">29</xref>].</p>
<p>It is also worth noting that large-scale resources not specifically designed for logical reasoning have still played an enabling role in benchmark development and model training. For example, YouTube-8M [<xref ref-type="bibr" rid="ref-86">86</xref>] provided large-scale video infrastructure that later supported progress in video-language understanding and temporal reasoning. Although such resources do not directly evaluate logical inference, they contributed to the broader ecosystem in which reasoning-oriented multimodal systems emerged.</p>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Evaluation Metrics</title>
<p>Evaluating complex logical reasoning in LVLMs remains fundamentally challenging because correct answers do not necessarily imply valid reasoning processes [<xref ref-type="bibr" rid="ref-125">125</xref>,<xref ref-type="bibr" rid="ref-126">126</xref>]. Existing evaluation protocols can be broadly divided into <italic>answer-level</italic> metrics and <italic>process-level</italic> metrics, each capturing a different aspect of model behavior. <xref ref-type="fig" rid="fig-11">Fig. 11</xref> summarizes this distinction between answer-level and process-level evaluation.</p>
<fig id="fig-11">
<label>Figure 11</label>
<caption>
<title>Evaluation metrics for complex logical reasoning in LVLMs, including Answer-level metrics and Process-level metrics. However, process-level evaluation is harder to scale because it often requires annotated reasoning traces, verifier models, or controlled perturbation tests.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-11.tif"/>
</fig>
<p><bold>Answer-level metrics.</bold> Answer-level metrics evaluate whether the final prediction matches a reference answer, without explicitly considering how that answer was obtained. Common examples include <italic>accuracy</italic>, <italic>F1 score</italic>, and <italic>exact match</italic>. These metrics remain widely used because they are simple, scalable, and easy to compare across models and benchmarks.</p>
<p>However, their limitations are especially pronounced in logical reasoning tasks. First, answer-level scores cannot distinguish between <italic>genuine inference</italic> and <italic>shortcut-based prediction</italic>. A model may produce the correct answer by exploiting dataset regularities or shallow multimodal cues rather than by carrying out the intended reasoning process. Second, answer-only evaluation cannot represent the existence of multiple valid reasoning paths, which is common in open-ended or multi-step tasks. Third, strict surface-form matching can be brittle: two answers may be semantically equivalent while differing in wording or decomposition.</p>
<p><bold>Process-level metrics.</bold> To address these weaknesses, process-level metrics attempt to evaluate the quality, validity, and structure of intermediate reasoning steps [<xref ref-type="bibr" rid="ref-124">124</xref>,<xref ref-type="bibr" rid="ref-126">126</xref>].</p>
<p>These dimensions can be organized as a structured evaluation protocol:<list list-type="simple">
<list-item>
<label>1.</label>
<p><italic>Step correctness</italic>: verifying whether each intermediate inference step is logically valid;</p></list-item>
<list-item>
<label>2.</label>
<p><italic>Faithfulness</italic>: assessing whether reasoning steps are causally grounded in the input rather than post-hoc rationalization;</p></list-item>
<list-item>
<label>3.</label>
<p><italic>Consistency</italic>: testing robustness under paraphrased or semantically equivalent inputs;</p></list-item>
<list-item>
<label>4.</label>
<p><italic>Completeness</italic>: ensuring that all necessary reasoning components are included in the inference chain.</p></list-item>
</list></p>
<p>Despite their conceptual appeal, process-level metrics are difficult to deploy at scale. They often require detailed annotations of intermediate steps or access to internal model traces, both of which are expensive and difficult to standardize across tasks [<xref ref-type="bibr" rid="ref-125">125</xref>,<xref ref-type="bibr" rid="ref-126">126</xref>]. Moreover, ground-truth reasoning trajectories are often not unique: different but equally valid solution paths may exist for the same problem [<xref ref-type="bibr" rid="ref-124">124</xref>]. This makes process supervision and evaluation substantially harder than answer-only scoring.</p>
<p>More fundamentally, even current process-level metrics do not fully capture several key aspects of multimodal logical reasoning, including cross-modal grounding validity, abstraction consistency, and causal coherence [<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>]. For this reason, there remains a strong need for evaluation frameworks that jointly assess <italic>correctness</italic>, <italic>faithfulness</italic>, and <italic>generalization</italic> under realistic multimodal conditions [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-46">46</xref>].</p>
<p>To make the discussion of empirical findings more explicit, we further summarize the main patterns observed across existing evaluations. Current LVLMs generally show stronger performance on perception-oriented and short-answer multimodal tasks, but their performance becomes less stable when tasks require multi-step inference, cross-modal evidence integration, counterfactual reasoning, or faithful explanation generation. Therefore, we emphasize not only reported performance scores but also the evaluation conditions, reasoning requirements, and failure patterns behind these results.</p>
</sec>
<sec id="s5_3">
<label>5.3</label>
<title>Cross-Method Quantitative Comparison and Standardized Evaluation Criteria</title>
<p>Although existing studies report substantial progress in LVLM reasoning, direct cross-method comparison remains difficult because different works adopt different benchmarks, answer formats, decoding settings, and modality assumptions [<xref ref-type="bibr" rid="ref-2">2</xref>,<xref ref-type="bibr" rid="ref-3">3</xref>,<xref ref-type="bibr" rid="ref-50">50</xref>]. To make the comparison more systematic, we further conduct a standardized quantitative synthesis over representative LVLMs and reasoning-oriented benchmarks. Rather than treating heterogeneous benchmark results as fully interchangeable, we organize the comparison along three reasoning-relevant axes: (1) single-image multimodal reasoning, measured by MathVista [<xref ref-type="bibr" rid="ref-78">78</xref>], MMBench [<xref ref-type="bibr" rid="ref-45">45</xref>], and MMMU [<xref ref-type="bibr" rid="ref-31">31</xref>]; (2) multi-image reasoning, measured by MuirBench [<xref ref-type="bibr" rid="ref-132">132</xref>]; and (3) video temporal reasoning, measured by Video-MME [<xref ref-type="bibr" rid="ref-131">131</xref>]. These benchmarks are selected because they are widely used in recent LVLM evaluation and provide explicit quantitative results for representative models [<xref ref-type="bibr" rid="ref-4">4</xref>,<xref ref-type="bibr" rid="ref-30">30</xref>].</p>
<p>We adopt the following standardization criteria. First, we only include numerical results that are explicitly reported in the original papers or official benchmark pages [<xref ref-type="bibr" rid="ref-7">7</xref>,<xref ref-type="bibr" rid="ref-37">37</xref>,<xref ref-type="bibr" rid="ref-95">95</xref>]. Missing results are marked as &#x201C;&#x2013;&#x201D; and are not imputed. Second, all reported values are accuracy scores in percentage form unless otherwise noted. Third, for the single-image reasoning axis, we compute a simple descriptive average over MathVista, MMBench, and MMMU only when all three values are available [<xref ref-type="bibr" rid="ref-19">19</xref>,<xref ref-type="bibr" rid="ref-26">26</xref>].<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:msub><mml:mrow><mml:mtext>R-Avg</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>SI</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>3</mml:mn></mml:mfrac><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mrow><mml:mtext>MathVista</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mrow><mml:mtext>MMBench</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mrow><mml:mtext>MMMU</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:msub><mml:mi>A</mml:mi><mml:mi>b</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> denotes the accuracy of model <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mi>m</mml:mi></mml:math></inline-formula> on benchmark <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mi>b</mml:mi></mml:math></inline-formula>. This quantity is intended only as a coarse descriptive index, not as a statistically rigorous meta-analytic estimate, because MathVista, MMBench, and MMMU differ in domain mix, answer format, and evaluation protocol. Fourth, for Video-MME, we report the &#x201C;without subtitles&#x201D; setting when both subtitle-free and subtitle-enhanced results are available, because it better isolates visual-temporal reasoning from additional textual transcript information. Finally, we do not collapse single-image, multi-image, and video scores into a single overall score, since these axes measure different reasoning conditions and should not be interpreted as strictly equivalent. Open-weight and proprietary systems are also listed separately in <xref ref-type="table" rid="table-5">Table 5</xref>, because their training data, system prompts, and evaluation interfaces are not fully standardized.</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Standardized quantitative comparison of representative LVLMs on reasoning-relevant benchmarks. All numbers are accuracy scores (%). &#x201C;&#x2013;&#x201D; indicates that no directly comparable value was found in the surveyed sources. <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:msub><mml:mrow><mml:mi mathvariant="normal">R</mml:mi><mml:mstyle displaystyle="false" scriptlevel="0"><mml:mtext>-</mml:mtext></mml:mstyle><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">v</mml:mi><mml:mi mathvariant="normal">g</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">S</mml:mi><mml:mi mathvariant="normal">I</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula> is used only as a descriptive single-image summary when MathVista, MMBench, and MMMU are all reported. Open-weight and proprietary models are listed separately because their system boundaries and training conditions are not fully standardized. For Video-MME, we report the subtitle-free setting.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>Model/Method</th>
<th>Main Design Paradigm</th>
<th>MathVista</th>
<th>MMBench</th>
<th>MMMU</th>
<th><inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:msub><mml:mrow><mml:mi mathvariant="normal">R</mml:mi><mml:mstyle displaystyle="false" scriptlevel="0"><mml:mtext>-</mml:mtext></mml:mstyle><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">v</mml:mi><mml:mi mathvariant="normal">g</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">S</mml:mi><mml:mi mathvariant="normal">I</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula></th>
<th>MuirBench</th>
<th>Video-MME</th>
</tr>
</thead>
<tbody>
<tr>
<td align="center" colspan="8"><italic>Open-weight/publicly described systems</italic></td>
</tr>
<tr>
<td>Qwen-VL-Max [<xref ref-type="bibr" rid="ref-7">7</xref>]</td>
<td>Unified LVLM/visual-language alignment</td>
<td>51.0</td>
<td>77.6</td>
<td>51.4</td>
<td>60.0</td>
<td>&#x2013;</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>InternVL2-8B [<xref ref-type="bibr" rid="ref-95">95</xref>]</td>
<td>Scaled vision backbone and alignment</td>
<td>58.3</td>
<td>81.7</td>
<td>49.3</td>
<td>63.1</td>
<td>&#x2013;</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>InternVL2-26B [<xref ref-type="bibr" rid="ref-95">95</xref>]</td>
<td>Scaled vision backbone and alignment</td>
<td>59.4</td>
<td>83.4</td>
<td>48.3</td>
<td>63.7</td>
<td>&#x2013;</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>LLaVA-OneVision-0.5B [<xref ref-type="bibr" rid="ref-100">100</xref>]</td>
<td>Unified multi-scenario transfer</td>
<td>34.8</td>
<td>52.1</td>
<td>31.4</td>
<td>39.4</td>
<td>25.5</td>
<td>44.0</td>
</tr>
<tr>
<td>LLaVA-OneVision-7B [<xref ref-type="bibr" rid="ref-100">100</xref>]</td>
<td>Unified multi-scenario transfer</td>
<td>63.2</td>
<td>80.8</td>
<td>48.8</td>
<td>64.3</td>
<td>41.8</td>
<td>58.2</td>
</tr>
<tr>
<td>LLaVA-OneVision-72B [<xref ref-type="bibr" rid="ref-100">100</xref>]</td>
<td>Unified multi-scenario transfer</td>
<td>67.5</td>
<td>85.9</td>
<td>56.8</td>
<td>70.1</td>
<td>54.8</td>
<td>66.2</td>
</tr>
<tr>
<td align="center" colspan="8"><italic>Proprietary systems</italic></td> 
</tr>
<tr>
<td>GPT-4V [<xref ref-type="bibr" rid="ref-37">37</xref>]</td>
<td>Proprietary LVLM</td>
<td>49.9</td>
<td>75.0</td>
<td>56.8</td>
<td>60.6</td>
<td>62.3</td>
<td>59.9</td>
</tr>
<tr>
<td>GPT-4o [<xref ref-type="bibr" rid="ref-37">37</xref>]</td>
<td>Proprietary omni-modal LVLM</td>
<td>63.8</td>
<td>&#x2013;</td>
<td>69.1</td>
<td>&#x2013;</td>
<td>&#x2013;</td>
<td>71.9</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="table" rid="table-5">Table 5</xref> and <xref ref-type="fig" rid="fig-12">Fig. 12</xref> summarize the resulting comparison. The results show several recurring trends. First, model scaling improves reasoning performance, as seen from the LLaVA-OneVision series: <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:msub><mml:mrow><mml:mi mathvariant="normal">R</mml:mi><mml:mstyle displaystyle="false" scriptlevel="0"><mml:mtext>-</mml:mtext></mml:mstyle><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">v</mml:mi><mml:mi mathvariant="normal">g</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">S</mml:mi><mml:mi mathvariant="normal">I</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula> increases from 39.4 for the 0.5B model to 70.1 for the 72B model, while MuirBench [<xref ref-type="bibr" rid="ref-132">132</xref>] and Video-MME [<xref ref-type="bibr" rid="ref-131">131</xref>] also improve consistently. Second, strong single-image reasoning does not automatically imply equally strong multi-image or video reasoning. For example, GPT-4V obtains a competitive single-image descriptive average [<xref ref-type="bibr" rid="ref-37">37</xref>,<xref ref-type="bibr" rid="ref-47">47</xref>], but LLaVA-OneVision-72B [<xref ref-type="bibr" rid="ref-100">100</xref>] reports stronger Video-MME performance in the subtitle-free setting. Third, the multi-image setting remains particularly challenging: even strong models show a noticeable drop on MuirBench compared with standard single-image benchmarks such as MMBench [<xref ref-type="bibr" rid="ref-45">45</xref>] and MMMU [<xref ref-type="bibr" rid="ref-31">31</xref>]. These observations should be interpreted within access-matched groups or as broad evidence trends rather than as a fully controlled ranking across all open and proprietary systems.</p>
<fig id="fig-12">
<label>Figure 12</label>
<caption>
<title>Multi-axis quantitative comparison of representative LVLMs. Single-image <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:msub><mml:mrow><mml:mi mathvariant="normal">R</mml:mi><mml:mstyle displaystyle="false" scriptlevel="0"><mml:mtext>-</mml:mtext></mml:mstyle><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">v</mml:mi><mml:mi mathvariant="normal">g</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">S</mml:mi><mml:mi mathvariant="normal">I</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula> is a descriptive average of MathVista, MMBench, and MMMU when all three are reported. MuirBench evaluates robust multi-image reasoning, while Video-MME evaluates video understanding under the subtitle-free setting. The comparison is intended as an evidence map rather than a fully controlled ranking, and it shows that improvements on single-image reasoning do not necessarily transfer uniformly to multi-image and video reasoning.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-12.tif"/>
</fig>
<p>The quantitative comparison also clarifies the limitations of current evidence. Some reasoning mechanisms reviewed in <xref ref-type="sec" rid="s4">Section 4</xref>, such as program-of-thought [<xref ref-type="bibr" rid="ref-43">43</xref>], neuro-symbolic reasoning [<xref ref-type="bibr" rid="ref-49">49</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>], and self-correction [<xref ref-type="bibr" rid="ref-44">44</xref>], often report results on task-specific benchmarks rather than on a shared multimodal reasoning suite. For instance, MathVista [<xref ref-type="bibr" rid="ref-78">78</xref>] reports tool-augmented textual baselines such as PoT GPT-4 with caption and OCR inputs, but these results are not directly equivalent to end-to-end LVLM inference because the visual information has already been converted into text [<xref ref-type="bibr" rid="ref-26">26</xref>]. Therefore, instead of claiming a universal ranking across all reasoning mechanisms, we treat <xref ref-type="table" rid="table-5">Table 5</xref> as a conservative evidence map: it compares representative systems only where public and directly comparable benchmark scores are available. This analysis reinforces the need for future standardized evaluation protocols that jointly report final accuracy, reasoning-process validity, modality-specific grounding, and robustness under distribution shift [<xref ref-type="bibr" rid="ref-13">13</xref>,<xref ref-type="bibr" rid="ref-19">19</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>].</p>

</sec>
<sec id="s5_4">
<label>5.4</label>
<title>Limitations of Current Evaluation</title>
<p>Despite the growing number of multimodal benchmarks, existing evaluation protocols remain insufficient for rigorously assessing complex logical reasoning in LVLMs [<xref ref-type="bibr" rid="ref-25">25</xref>,<xref ref-type="bibr" rid="ref-29">29</xref>]. <xref ref-type="fig" rid="fig-13">Fig. 13</xref> summarizes the major limitations of current LVLM evaluation protocols. Existing benchmarks mainly emphasize final-answer accuracy, but this metric alone cannot reveal whether the model follows a valid reasoning process, grounds its answer in visual evidence, or remains robust under distribution shift and adversarial perturbations.</p>
<fig id="fig-13">
<label>Figure 13</label>
<caption>
<title>Main limitations of current LVLM evaluation protocols. Existing benchmarks often emphasize final-answer accuracy, while providing limited evidence about reasoning-process validity, visual grounding, and robustness under challenging conditions.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-13.tif"/>
</fig>
<p><bold>Benchmark contamination and data leakage.</bold> A fundamental concern is that benchmark instances may appear, directly or indirectly, in large pretraining corpora [<xref ref-type="bibr" rid="ref-37">37</xref>]. Because many popular benchmarks are publicly available, large-scale models may partially memorize patterns or examples from them, which can inflate apparent reasoning performance. This is especially problematic for logical reasoning, where benchmark success may otherwise be misinterpreted as evidence of genuine inferential capability [<xref ref-type="bibr" rid="ref-135">135</xref>].</p>
<p><bold>Limited coverage of reasoning types.</bold> Most existing benchmarks probe only a narrow subset of reasoning skills, such as visual pattern induction [<xref ref-type="bibr" rid="ref-22">22</xref>] or synthetic compositional reasoning [<xref ref-type="bibr" rid="ref-39">39</xref>]. Yet real multimodal reasoning often requires the joint use of multiple forms of inference, including deductive, abductive, causal, and compositional reasoning [<xref ref-type="bibr" rid="ref-9">9</xref>,<xref ref-type="bibr" rid="ref-23">23</xref>,<xref ref-type="bibr" rid="ref-75">75</xref>]. The fragmented nature of current benchmark design therefore makes comprehensive evaluation difficult.</p>
<p><bold>Outcome-oriented evaluation bias.</bold> A dominant limitation of current practice is the heavy reliance on final-answer correctness [<xref ref-type="bibr" rid="ref-44">44</xref>]. This biases evaluation toward outcomes rather than reasoning validity. As a result, models that exploit superficial multimodal regularities may appear competitive with models that genuinely perform structured inference. Existing benchmarks are therefore often insufficient to separate <italic>true reasoning</italic> from <italic>shortcut-based inference</italic> [<xref ref-type="bibr" rid="ref-1">1</xref>].</p>
<p><bold>Synthetic-to-real gap.</bold> A substantial portion of existing reasoning benchmarks is based on synthetic data [<xref ref-type="bibr" rid="ref-22">22</xref>,<xref ref-type="bibr" rid="ref-39">39</xref>]. Synthetic settings are useful because they provide control and reduce annotation ambiguity, but they lack the noise, ambiguity, and long-tail variability of real-world multimodal inputs. Models that perform well in controlled synthetic environments may therefore fail to transfer their reasoning ability to realistic scenarios [<xref ref-type="bibr" rid="ref-31">31</xref>].</p>
<p><bold>Scale and statistical reliability.</bold> Many logical reasoning benchmarks are relatively small, limiting their power to support statistically stable model comparison [<xref ref-type="bibr" rid="ref-30">30</xref>,<xref ref-type="bibr" rid="ref-45">45</xref>]. Small test sets are more vulnerable to variance, narrow coverage, and benchmark-specific artifacts, making it difficult to draw reliable conclusions about general reasoning ability.</p>
<p><bold>Distributional robustness and generalization.</bold> Performance on standard benchmarks often degrades substantially under distribution shift, including changes in visual style, wording, task format, or evidence layout [<xref ref-type="bibr" rid="ref-31">31</xref>,<xref ref-type="bibr" rid="ref-46">46</xref>]. This raises a broader concern: even when a model performs well on a benchmark, it may still lack the robustness required for practical reasoning deployment.</p>
<p><bold>Annotation quality and ambiguity.</bold> Evaluation is further complicated by annotation noise, underspecification, and the fact that many logical reasoning tasks admit multiple plausible solution paths [<xref ref-type="bibr" rid="ref-126">126</xref>]. In such cases, defining a single gold answer or a single gold reasoning trace may be inherently reductive, which complicates fair comparison across systems.</p>
</sec>
</sec>
<sec id="s6">
<label>6</label>
<title>Challenges and Open Problems</title>
<p>Rather than repeating earlier descriptive comparisons, this section distills the main unresolved <italic>technical bottlenecks</italic> that recur across architectures, mechanisms, datasets, and evaluation settings. We focus on five cross-cutting issues: process validity, long-horizon state management, multimodal grounding, data quality, and evaluation realism. The goal is not to restate every earlier limitation, but to synthesize the recurring failure patterns that most consistently prevent current LVLMs from becoming reliable reasoning systems.</p>
<sec id="s6_1">
<label>6.1</label>
<title>Reasoning Faithfulness and Validity</title>
<p>A central challenge is ensuring that model outputs reflect genuine logical inference rather than superficial pattern matching [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-44">44</xref>]. Even when LVLMs produce correct answers, those answers may be driven by spurious correlations, memorized shortcuts, or dataset regularities rather than valid reasoning [<xref ref-type="bibr" rid="ref-135">135</xref>].</p>
<p>This faithfulness problem has several sources. First, training data may contain exploitable biases, allowing models to succeed without learning the intended reasoning patterns [<xref ref-type="bibr" rid="ref-33">33</xref>]. Second, current benchmarks often emphasize final-answer accuracy and therefore do not clearly distinguish between genuine inference and shallow answer prediction [<xref ref-type="bibr" rid="ref-29">29</xref>]. Third, many reasoning tasks involve implicit premises, background assumptions, or underspecified constraints, making it difficult to determine whether a model is reasoning correctly or simply generating plausible outputs [<xref ref-type="bibr" rid="ref-62">62</xref>].</p>
<p>Recent work on neuro-symbolic reasoning and reasoning robustness has further emphasized that answer correctness alone is insufficient evidence of valid reasoning [<xref ref-type="bibr" rid="ref-49">49</xref>]. Addressing this challenge requires training objectives and evaluation protocols that target faithfulness directly, rather than treating it as a by-product of answer accuracy.</p>
</sec>
<sec id="s6_2">
<label>6.2</label>
<title>Hallucination in Reasoning</title>
<p>Hallucination remains a major challenge in LVLM reasoning, but in reasoning settings it is especially problematic because errors can compound across inference steps [<xref ref-type="bibr" rid="ref-59">59</xref>,<xref ref-type="bibr" rid="ref-60">60</xref>]. A hallucinated intermediate conclusion may appear coherent and even support later steps, ultimately leading to a convincing but incorrect final answer.</p>
<p>This problem is more difficult than ordinary generation errors because reasoning hallucinations are often structurally plausible. Models may invent unsupported relations, attributes, or causal links, and such errors can remain hidden inside a multi-step chain [<xref ref-type="bibr" rid="ref-90">90</xref>]. The risk increases when reasoning chains become longer, when external grounding is weak, or when the model must infer latent structure from noisy visual evidence [<xref ref-type="bibr" rid="ref-44">44</xref>,<xref ref-type="bibr" rid="ref-57">57</xref>].</p>
<p>Reducing reasoning hallucination therefore requires more than improving fluent generation. It requires stronger grounding, intermediate verification, and mechanisms for checking whether each step is supported by the input or by trusted external knowledge [<xref ref-type="bibr" rid="ref-82">82</xref>].</p>
</sec>
<sec id="s6_3">
<label>6.3</label>
<title>Long-Range Reasoning Capabilities</title>
<p>Complex logical reasoning often requires maintaining coherence across long inference chains, yet long-range reasoning remains a persistent weakness of current models [<xref ref-type="bibr" rid="ref-57">57</xref>,<xref ref-type="bibr" rid="ref-121">121</xref>]. As the number of reasoning steps increases, models must preserve relevant intermediate states, retrieve earlier premises, and prevent local errors from cascading through later steps.</p>
<p>Current LVLMs still struggle with these requirements. Long reasoning chains place pressure on memory, attention allocation, and step-by-step consistency [<xref ref-type="bibr" rid="ref-32">32</xref>]. Earlier mistakes can easily propagate, and models may lose track of previously established constraints or fail to integrate them correctly into later stages [<xref ref-type="bibr" rid="ref-44">44</xref>,<xref ref-type="bibr" rid="ref-116">116</xref>]. These issues are particularly severe in multi-image, long-document, and video settings, where evidence is distributed across extended multimodal contexts.</p>
<p>Improving long-range reasoning will likely require better memory mechanisms, more reliable intermediate-state management, and architectures or training strategies that explicitly support long-horizon inference.</p>
</sec>
<sec id="s6_4">
<label>6.4</label>
<title>Cross-Modal Representation Alignment</title>
<p>Robust logical reasoning in LVLMs depends on accurate alignment between visual and textual representations [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-52">52</xref>]. This remains difficult because the two modalities differ substantially in form: visual information is continuous, spatial, and often ambiguous, whereas textual information is discrete, sequential, and usually more explicit [<xref ref-type="bibr" rid="ref-7">7</xref>].</p>
<p>For reasoning tasks, alignment must go beyond coarse semantic matching. Models must determine which visual evidence corresponds to which textual premise, maintain consistency across modalities, and resolve conflicts when the two sources provide incomplete or partially inconsistent information [<xref ref-type="bibr" rid="ref-45">45</xref>,<xref ref-type="bibr" rid="ref-111">111</xref>]. Misalignment at this stage can undermine the entire reasoning process, even if the downstream inference mechanism is otherwise strong [<xref ref-type="bibr" rid="ref-95">95</xref>].</p>
<p>This challenge suggests that future work should not treat representation alignment as merely a perception problem. Instead, alignment mechanisms must be designed with reasoning in mind, especially for tasks involving compositional structure, abstraction, and multimodal constraint satisfaction.</p>
</sec>
<sec id="s6_5">
<label>6.5</label>
<title>Data Quality and Scarcity</title>
<p>The development of logical reasoning in LVLMs is constrained not only by limited data volume, but also by the quality of the reasoning data itself [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-26">26</xref>]. In contrast to general multimodal instruction data, reasoning-oriented corpora must often include carefully designed problem structures, valid intermediate reasoning steps, reliable annotations of latent relations, and enough diversity to prevent shortcut learning. These requirements make high-quality data expensive to construct and difficult to scale.</p>
<p>This challenge has several dimensions. First, annotating reasoning trajectories requires expertise and is often labor-intensive [<xref ref-type="bibr" rid="ref-64">64</xref>]. Second, synthetic data can help scale reasoning supervision, but may not transfer well to realistic multimodal settings [<xref ref-type="bibr" rid="ref-22">22</xref>,<xref ref-type="bibr" rid="ref-39">39</xref>]. Third, dataset quality is highly sensitive to hidden sampling bias, annotation regularity, and source imbalance. Seemingly broad datasets may still overrepresent particular domains, visual layouts, answer formats, or reasoning templates, which can allow models to exploit superficial regularities instead of learning transferable inference patterns [<xref ref-type="bibr" rid="ref-31">31</xref>,<xref ref-type="bibr" rid="ref-39">39</xref>,<xref ref-type="bibr" rid="ref-45">45</xref>]. In multimodal reasoning, such bias may appear as shortcut dependence on OCR artifacts, visual co-occurrence priors, or benchmark-specific answer distributions.</p>
<p>Data quality problems also interact with evaluation quality. If benchmark construction reuses narrow templates, weak negative cases, or partially contaminated sources, answer accuracy can overestimate genuine reasoning ability [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-48">48</xref>]. For this reason, better reasoning data should be judged not only by scale, but also by diversity, annotation reliability, source balance, and resistance to shortcut exploitation.</p>
<p>As a result, progress in reasoning often depends on relatively small or specialized datasets, which is unlikely to be sufficient for building broad, robust, and transferable multimodal reasoning capabilities. Better data construction pipelines, scalable synthetic-real mixtures, stronger annotation audits, and higher-quality process supervision remain important open problems.</p>
</sec>
<sec id="s6_6">
<label>6.6</label>
<title>Evaluation Gaps</title>
<p>Finally, current benchmarks and evaluation protocols still provide only a partial view of reasoning ability [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-134">134</xref>]. Benchmark performance may not reliably reflect whether a model can reason in a grounded, generalizable, and faithful manner [<xref ref-type="bibr" rid="ref-31">31</xref>,<xref ref-type="bibr" rid="ref-45">45</xref>].</p>
<p>Several issues contribute to this gap. Existing benchmarks often cover only a limited subset of reasoning types [<xref ref-type="bibr" rid="ref-35">35</xref>]; many are susceptible to contamination or data leakage [<xref ref-type="bibr" rid="ref-48">48</xref>]; and most still emphasize answer-level outcomes over reasoning processes [<xref ref-type="bibr" rid="ref-120">120</xref>]. As discussed in the previous section, this makes it difficult to distinguish true reasoning from benchmark-specific shortcuts.</p>
<p>For this reason, evaluation remains one of the most important unresolved problems in the field. Progress will depend on benchmarks that cover a broader range of reasoning skills, are more robust to contamination and distribution shift, and include process-aware protocols capable of assessing not only correctness, but also faithfulness and reasoning quality.</p>
<p>Taken together, these bottlenecks show that the central obstacle is no longer raw benchmark coverage alone, but the difficulty of maintaining grounded, verifiable, and transferable reasoning under realistic multimodal conditions. <xref ref-type="table" rid="table-6">Table 6</xref> summarizes the main challenge-to-evidence mapping discussed in this section and highlights the most actionable open directions [<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>].</p>
<table-wrap id="table-6">
<label>Table 6</label>
<caption>
<title>Challenge-to-evidence mapping in recent LVLM reasoning research.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>Challenge</th>
<th>Representative Evidence</th>
<th>Main Observation</th>
<th>Open Direction</th>
</tr>
</thead>
<tbody>
<tr>
<td>Faithfulness gap</td>
<td>Neuro-symbolic reliability analyses [<xref ref-type="bibr" rid="ref-49">49</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>], Multimodal CoT studies [<xref ref-type="bibr" rid="ref-116">116</xref>,<xref ref-type="bibr" rid="ref-120">120</xref>]</td>
<td>Correct final answers can still arise from brittle shortcuts, implicit heuristics, or post-hoc rationalizations rather than truly grounded intermediate reasoning</td>
<td>Process-grounded supervision, verifier-guided decoding, and causally faithful reasoning traces</td>
</tr>
<tr>
<td>Evaluation saturation</td>
<td>Harder benchmarks such as MMMU-Pro and MuirBench [<xref ref-type="bibr" rid="ref-31">31</xref>,<xref ref-type="bibr" rid="ref-132">132</xref>], broader benchmark analyses [<xref ref-type="bibr" rid="ref-25">25</xref>,<xref ref-type="bibr" rid="ref-29">29</xref>]</td>
<td>Many earlier benchmarks are becoming less discriminative, making it difficult to separate shallow answer matching from robust multimodal reasoning ability</td>
<td>Continually updated hard-split benchmarks, adversarial benchmark construction, and process-sensitive evaluation</td>
</tr>
<tr>
<td>Temporal reasoning bottleneck</td>
<td>Video-MME [<xref ref-type="bibr" rid="ref-131">131</xref>], VCR-Bench [<xref ref-type="bibr" rid="ref-134">134</xref>], video QA benchmarks such as TVQA and ActivityNet-QA [<xref ref-type="bibr" rid="ref-83">83</xref>,<xref ref-type="bibr" rid="ref-84">84</xref>]</td>
<td>LVLMs remain substantially weaker on long-horizon temporal dependency modeling than on static image reasoning, especially when evidence must be aggregated across distant frames</td>
<td>Memory-aware temporal reasoning, event-structured representations, and long-context video planning mechanisms</td>
</tr>
<tr>
<td>Math-visual reasoning instability</td>
<td>MathVista [<xref ref-type="bibr" rid="ref-78">78</xref>], MathVerse [<xref ref-type="bibr" rid="ref-80">80</xref>], OlympiadBench [<xref ref-type="bibr" rid="ref-27">27</xref>], UniGeo [<xref ref-type="bibr" rid="ref-122">122</xref>]</td>
<td>Models frequently fail in symbolic precision, formula grounding, diagram interpretation, and multi-step numerical consistency even when coarse semantic understanding is correct</td>
<td>Hybrid symbolic tool use, theorem-aware solvers, and explicit visual-to-formal representation interfaces</td>
</tr>
<tr>
<td>Cross-scene aggregation weakness</td>
<td>LLaVA-OneVision [<xref ref-type="bibr" rid="ref-100">100</xref>], MuirBench [<xref ref-type="bibr" rid="ref-132">132</xref>], LLaVA-NeXT-Interleave [<xref ref-type="bibr" rid="ref-72">72</xref>]</td>
<td>Evidence fusion across multiple images or interleaved contexts is markedly less stable than single-image inference, with errors in entity alignment and cross-image comparison</td>
<td>Structured multi-image planning, entity-centric memory, and graph-based cross-scene reasoning</td>
</tr>
<tr>
<td>Spatial reasoning fragility</td>
<td>Mind the Gap [<xref ref-type="bibr" rid="ref-87">87</xref>], Think3D [<xref ref-type="bibr" rid="ref-115">115</xref>], EmbodiedVSR [<xref ref-type="bibr" rid="ref-74">74</xref>], 3D-R1 [<xref ref-type="bibr" rid="ref-136">136</xref>]</td>
<td>Spatial relations, viewpoint changes, occlusion, depth cues, and 3D consistency remain persistent failure points even for otherwise strong general-purpose LVLMs</td>
<td>Geometry-aware intermediate representations, 3D-augmented supervision, and embodied spatial reasoning frameworks</td>
</tr>
<tr>
<td>Causal reasoning deficiency</td>
<td>CausalVLBench [<xref ref-type="bibr" rid="ref-34">34</xref>], Multimodal Causal Reasoning Benchmark/MuCR/VcCoT [<xref ref-type="bibr" rid="ref-10">10</xref>], CausalBench [<xref ref-type="bibr" rid="ref-35">35</xref>]</td>
<td>Current models often capture correlation patterns rather than genuine causal structure, leading to weak performance on intervention, counterfactual, and cause-effect attribution tasks</td>
<td>Interventional data construction, causal representation learning, and counterfactual multimodal reasoning pipelines</td>
</tr>
<tr>
<td>Inductive generalization weakness</td>
<td>RAVEN [<xref ref-type="bibr" rid="ref-22">22</xref>], InPhyRe [<xref ref-type="bibr" rid="ref-67">67</xref>], MM-IQ [<xref ref-type="bibr" rid="ref-46">46</xref>], inductive multimodal learning studies [<xref ref-type="bibr" rid="ref-66">66</xref>]</td>
<td>Models still struggle to abstract reusable rules from limited observations, particularly when test distributions differ from training patterns or require physical/general relational induction</td>
<td>Rule-centric pretraining, abstraction-oriented curricula, and out-of-distribution inductive evaluation</td>
</tr>
<tr>
<td>Abductive reasoning uncertainty</td>
<td>Reasoner [<xref ref-type="bibr" rid="ref-23">23</xref>], DixitWorld [<xref ref-type="bibr" rid="ref-75">75</xref>], MAR [<xref ref-type="bibr" rid="ref-9">9</xref>]</td>
<td>Inferring the most plausible hidden explanation from incomplete multimodal evidence remains difficult, especially when several explanations are superficially compatible</td>
<td>Hypothesis ranking objectives, uncertainty-aware decoding, and explanation-calibrated training</td>
</tr>
<tr>
<td>Grounding and hallucination tension</td>
<td>Hallucination analyses such as DHCP and CLAIM [<xref ref-type="bibr" rid="ref-59">59</xref>,<xref ref-type="bibr" rid="ref-60">60</xref>], broad LVLM evaluations [<xref ref-type="bibr" rid="ref-30">30</xref>,<xref ref-type="bibr" rid="ref-45">45</xref>]</td>
<td>Strong fluent generation can mask weak visual grounding, causing unsupported object mentions, attribute errors, or multilingual hallucinations during reasoning</td>
<td>Grounding-aware decoding, cross-modal verification, and uncertainty-calibrated generation</td>
</tr>
<tr>
<td>Document and chart reasoning robustness</td>
<td>BRIDGE [<xref ref-type="bibr" rid="ref-55">55</xref>], ChartQA [<xref ref-type="bibr" rid="ref-130">130</xref>], DocVQA [<xref ref-type="bibr" rid="ref-129">129</xref>], WebSRC [<xref ref-type="bibr" rid="ref-137">137</xref>]</td>
<td>Performance drops substantially when reasoning depends on layout structure, dense OCR, long documents, or mixed visual-textual evidence distributed across pages or regions</td>
<td>Layout-aware memory, OCR-reasoning co-optimization, and hierarchical document planning</td>
</tr>
<tr>
<td>Tool-use reliability and execution brittleness</td>
<td>Chameleon [<xref ref-type="bibr" rid="ref-71">71</xref>], Toolformer [<xref ref-type="bibr" rid="ref-106">106</xref>], ToRA [<xref ref-type="bibr" rid="ref-42">42</xref>], PAL [<xref ref-type="bibr" rid="ref-105">105</xref>]</td>
<td>Tool-augmented reasoning can improve difficult tasks, but end-to-end reliability is limited by planning errors, incorrect tool calls, execution failures, and cascading mistakes</td>
<td>Tool-selection policies, execution verification, and adaptive planner-verifier architectures</td>
</tr>
<tr>
<td>Reasoning cost and scalability</td>
<td>InternVL2.5 [<xref ref-type="bibr" rid="ref-101">101</xref>], GLM-4.1V-Think/GLM-4.5V [<xref ref-type="bibr" rid="ref-102">102</xref>], Vision-R1 [<xref ref-type="bibr" rid="ref-70">70</xref>], WeThink [<xref ref-type="bibr" rid="ref-68">68</xref>]</td>
<td>Stronger reasoning often depends on larger models, longer chains, test-time scaling, or reinforcement learning, increasing training and inference cost substantially</td>
<td>Efficient reasoning distillation, adaptive compute allocation, and budget-aware test-time scaling</td>
</tr>
<tr>
<td>Benchmark-to-reality transfer gap</td>
<td>Survey and benchmark analyses [<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>,<xref ref-type="bibr" rid="ref-29">29</xref>], WebArena [<xref ref-type="bibr" rid="ref-91">91</xref>]</td>
<td>Gains on curated benchmarks do not always translate to open-world multimodal tasks, where perception noise, ambiguity, tool interaction, and long-tail cases are much more severe</td>
<td>Realistic interactive benchmarks, deployment-oriented evaluation, and robustness testing under distribution shift</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s7">
<label>7</label>
<title>Future Directions</title>
<p>Despite rapid progress, complex logical reasoning in LVLMs remains far from being reliably solved. To avoid overly general recommendations, we organize future research directions around concrete and actionable technical routes, including structured reasoning data construction, reasoning-centric architecture design, verifier-guided training, process-aware evaluation, and deployment-oriented robustness testing [<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>]. <xref ref-type="fig" rid="fig-14">Fig. 14</xref> organizes these future directions into five interconnected research dimensions.</p>
<fig id="fig-14">
<label>Figure 14</label>
<caption>
<title>Overview of future research directions for complex logical reasoning in LVLMs. We organize future opportunities and progress into five interconnected dimensions: data, model design, training, evaluation, and applications/deployment.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_83586-fig-14.tif"/>
</fig>
<p><bold>Data: Constructing Structured Reasoning Corpora with Grounded Process Supervision.</bold> Future progress will depend heavily on better reasoning-oriented data. Existing multimodal datasets are still limited in realism, diversity, and process supervision [<xref ref-type="bibr" rid="ref-31">31</xref>,<xref ref-type="bibr" rid="ref-45">45</xref>]. A key direction is therefore to construct datasets that more faithfully reflect real-world reasoning conditions, including ambiguity, incomplete information, noisy evidence, and cross-modal inconsistency. Another important direction is to move from answer-only supervision toward richer annotations of intermediate reasoning steps, grounded evidence, and alternative valid solution paths [<xref ref-type="bibr" rid="ref-32">32</xref>,<xref ref-type="bibr" rid="ref-120">120</xref>]. Synthetic data will likely remain important for scaling, but it should be combined with realistic distributions and better data curation to avoid overfitting to artificial patterns [<xref ref-type="bibr" rid="ref-39">39</xref>]. Curriculum-style data construction, in which reasoning complexity increases progressively, may further support the acquisition of compositional and long-horizon reasoning abilities [<xref ref-type="bibr" rid="ref-121">121</xref>].</p>
<p><bold>Model Design: Toward Reasoning-Centric Architectures.</bold> Most current LVLMs are still optimized primarily for multimodal perception, generation, and instruction following, rather than for structured reasoning itself [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-95">95</xref>]. A more actionable direction is to design LVLMs with explicit intermediate-state memory, modular perception-reasoning interfaces, and controllable reasoning flows, allowing models to store premises, update intermediate conclusions, and verify cross-modal consistency during multi-step inference [<xref ref-type="bibr" rid="ref-41">41</xref>,<xref ref-type="bibr" rid="ref-63">63</xref>,<xref ref-type="bibr" rid="ref-138">138</xref>]. Neuro-symbolic integration remains especially promising because it offers a potential balance between neural flexibility and symbolic rigor [<xref ref-type="bibr" rid="ref-51">51</xref>,<xref ref-type="bibr" rid="ref-62">62</xref>]. More broadly, future architectures may need to support not only static 2D understanding, but also richer settings such as long-context multimodal reasoning, embodied interaction, and 3D scene reasoning [<xref ref-type="bibr" rid="ref-136">136</xref>]. At the same time, architectural advances must remain computationally practical, since many current reasoning-oriented systems rely on expensive inference-time procedures [<xref ref-type="bibr" rid="ref-57">57</xref>].</p>
<p><bold>Training: Learning to Reason, Not Just to Predict.</bold> Training paradigms should incorporate explicit supervision and optimization signals for reasoning quality, including step-level correctness, evidence grounding, contradiction detection, and verifier-based reward signals, rather than relying only on final-answer accuracy [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-44">44</xref>]. While reasoning-focused fine-tuning can improve performance, future progress will likely require more explicit supervision over intermediate trajectories, grounded evidence usage, and error correction behavior [<xref ref-type="bibr" rid="ref-116">116</xref>]. Reinforcement learning is a particularly promising direction when combined with verifiers, reward models, or structured feedback that assess logical validity rather than only task success [<xref ref-type="bibr" rid="ref-125">125</xref>,<xref ref-type="bibr" rid="ref-127">127</xref>]. In addition, multi-task and transfer learning may help reasoning capabilities generalize across domains rather than remaining benchmark-specific [<xref ref-type="bibr" rid="ref-79">79</xref>]. A further underexplored direction is training-time diagnosis of recurrent reasoning failures, so that models can be improved through targeted correction rather than only broader scaling [<xref ref-type="bibr" rid="ref-126">126</xref>].</p>
<p><bold>Evaluation: Beyond Accuracy toward Process-Aware Assessment.</bold> Evaluation will remain a central bottleneck unless it becomes more process-aware and more robust to shortcut solutions [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-134">134</xref>]. Future evaluation protocols should report not only final accuracy, but also step correctness, evidence faithfulness, cross-modal grounding consistency, robustness under paraphrased or perturbed inputs, and contamination-resistant performance on newly constructed test splits [<xref ref-type="bibr" rid="ref-120">120</xref>]. Adversarial evaluation, distribution-shift testing, and contamination-resistant benchmark design will be essential for measuring whether models truly generalize [<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>]. Another important direction is the development of scalable evaluation protocols that can assess reasoning processes without requiring prohibitively expensive human annotations. More broadly, future evaluation should better reflect the properties that matter in practice: interpretability, robustness, and reliable multimodal inference under realistic conditions [<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-119">119</xref>].</p>
<p><bold>Applications and Deployment: From Controlled Tasks to Reliable Reasoning Systems.</bold> Ultimately, the value of logical reasoning in LVLMs should be evaluated under deployment-oriented conditions, such as noisy visual inputs, incomplete textual context, conflicting multimodal evidence, tool-use failures, and distribution shifts [<xref ref-type="bibr" rid="ref-12">12</xref>]. Promising directions include scientific discovery, multimodal decision support, educational systems with step-by-step explanations, and high-stakes domains such as medicine and law, where reasoning validity and explainability are especially important [<xref ref-type="bibr" rid="ref-15">15</xref>]. These application settings place stricter demands on reliability, transparency, and robustness than controlled benchmarks do [<xref ref-type="bibr" rid="ref-25">25</xref>]. They also highlight the need for models that can reason over heterogeneous evidence, justify their conclusions in grounded ways, and remain stable under uncertainty. In this sense, real-world deployment should not be viewed merely as an application endpoint, but also as an important driver of future research priorities.</p>
</sec>
<sec id="s8">
<label>8</label>
<title>Conclusion</title>
<p>In this survey, we present a structured and comprehensive synthesis of complex logical reasoning in LVLMs, covering conceptual foundations, architectural paradigms, reasoning mechanisms, evaluation protocols, and empirical findings. Our analysis highlights a fundamental tension in current LVLM research: although these models achieve strong performance on benchmark tasks, their reasoning processes are often implicit, fragile, and insufficiently grounded in multimodal evidence. Moreover, this survey identifies several key challenges that continue to limit progress, including insufficient reasoning faithfulness, difficulties in long-horizon inference, imperfect cross-modal alignment, and the lack of process-aware evaluation frameworks. Looking ahead, we argue that future research should move beyond performance-driven optimization and focus more explicitly on reasoning-oriented design. In addition, there is a clear need for more rigorous and scalable evaluation methodologies that can assess reasoning faithfulness, robustness, and generalization. Overall, we believe that addressing these challenges will enable LVLMs to move beyond strong perceptual performance toward reliable, interpretable, and generalizable multimodal reasoning systems. We hope this survey can provide both a conceptual foundation and a practical roadmap for advancing research in this important direction.</p>
</sec>
</body>
<back>
<ack>
<p>The background panel in <xref ref-type="fig" rid="fig-8">Fig. 8</xref> (excluding its main content) and the small icons in <xref ref-type="fig" rid="fig-2">Figs. 2</xref> and <xref ref-type="fig" rid="fig-3">3</xref> were created using AI-generated content from Gemini 3 pro.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>The authors received no specific funding for this study.</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>The authors confirm contributions to this paper as follows: Weiqiang Jin: Conceptualization, Methodology, Formal analysis, Writing&#x2014;original draft, Writing&#x2014;review &#x0026; editing, Project administration, Validation; Yang Liu: Data curation, Writing&#x2014;review &#x0026; editing, Investigation, Funding acquisition; Yang Gao: Investigation, Visualization, Writing&#x2014;review &#x0026; editing; Shixiang Tang: Writing&#x2014;review &#x0026; editing; Yanghao Zhou: Writing&#x2014;review &#x0026; editing; Jinhu Qi: Formal analysis, Writing&#x2014;review &#x0026; editing; Wentao Zhang: Writing&#x2014;review &#x0026; editing; Junli Wang: Writing&#x2014;review &#x0026; editing; Jing Gao: Writing&#x2014;review &#x0026; editing; Yue Ma: Writing&#x2014;review &#x0026; editing; Ziwei Zhang: Investigation, Supervision, Writing&#x2014;review &#x0026; editing; Biao Zhao: Investigation, Supervision, Conceptualization, Project administration. This work is conducted by the first author, Weiqiang Jin, during his research at Xi&#x2019;an Jiaotong University under the guidance of Prof. Ziwei Zhang and Prof. Biao Zhao. All authors reviewed and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>This work did not involve any associated code or datasets as it is primarily theoretical in nature. No data were generated or analyzed during this study.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest.</p>
</sec>

<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wei</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Schuurmans</surname> <given-names>D</given-names></string-name>, <string-name><surname>Bosma</surname> <given-names>M</given-names></string-name>, <string-name><surname>Ichter</surname> <given-names>B</given-names></string-name>, <string-name><surname>Xia</surname> <given-names>F</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Chain-of-thought prompting elicits reasoning in large language models</article-title>. <source>Adv Neural Inf Process Syst</source>. <year>2022</year>;<volume>35</volume>:<fpage>24824</fpage>&#x2013;<lpage>37</lpage>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Gan</surname> <given-names>W</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>PS</given-names></string-name></person-group>. <article-title>Multimodal large language models: a survey</article-title>. <comment>arXiv:2311.13165. 2023</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2311.13165</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Caffagni</surname> <given-names>D</given-names></string-name>, <string-name><surname>Cocchi</surname> <given-names>F</given-names></string-name>, <string-name><surname>Barsellotti</surname> <given-names>L</given-names></string-name>, <string-name><surname>Moratelli</surname> <given-names>N</given-names></string-name>, <string-name><surname>Sarto</surname> <given-names>S</given-names></string-name>, <string-name><surname>Baraldi</surname> <given-names>L</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>The revolution of multimodal large language models: a survey</article-title>. In: <conf-name>Proceedings of the Findings of the Association for Computational Linguistics ACL 2024; 2024 Aug 11&#x2013;16</conf-name>; <publisher-loc>Bangkok, Thailand</publisher-loc>. p. <fpage>13590</fpage>&#x2013;<lpage>618</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2024.findings-acl.807</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Liang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Hong</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Shang</surname> <given-names>P</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Fu</surname> <given-names>Q</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A survey of multimodel large language models</article-title>. In: <conf-name>Proceedings of the 3rd International Conference on Computer, Artificial Intelligence and Control Engineering; 2024 Jan 26&#x2013;28</conf-name>; <publisher-loc>Xi&#x2019;an, China</publisher-loc>. p. <fpage>405</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3672758.3672824</pub-id>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>YJ</given-names></string-name></person-group>. <article-title>Visual instruction tuning</article-title>. In: <conf-name>Proceedings of the 37th International Conference on Neural Information Processing Systems NIPS&#x2019;23; 2023 Dec 10&#x2013;16; New Orleans, LA, USA</conf-name>. <publisher-loc> Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc.</publisher-name>; <year>2023</year>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>LLaVA-NeXT: improved reasoning, OCR, and world knowledge; 2024 [cited 2026 Jan 1]</article-title>. Available from: <ext-link ext-link-type="uri" xlink:href="https://llava-vl.github.io/blog/2024-01-30-llava-next/">https://llava-vl.github.io/blog/2024-01-30-llava-next/</ext-link>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Bai</surname> <given-names>J</given-names></string-name>, <string-name><surname>Bai</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Tan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>P</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Qwen-VL: a versatile vision-language model for understanding, localization, text reading, and beyond</article-title>. <comment>arXiv:2308.12966. 2023</comment>. doi<pub-id pub-id-type="doi">10.48550/arXiv.2308.12966</pub-id>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Agrawal</surname> <given-names>A</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Antol</surname> <given-names>S</given-names></string-name>, <string-name><surname>Mitchell</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zitnick</surname> <given-names>CL</given-names></string-name>, <string-name><surname>Parikh</surname> <given-names>D</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>VQA: visual question answering</article-title>. <source>Int J Comput Vis</source>. <year>2017</year>;<volume>123</volume>(<issue>1</issue>):<fpage>4</fpage>&#x2013;<lpage>31</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s11263-016-0966-6</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>M</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>T</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Han</surname> <given-names>K</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>Z</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Multi-modal action chain abductive reasoning</article-title>. In: <conf-name>Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers); 2023 Jul 9&#x2013;14</conf-name>; <publisher-loc>Toronto, ON, Canada</publisher-loc>. p. <fpage>4617</fpage>&#x2013;<lpage>28</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2023.acl-long.254</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>D</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>A</given-names></string-name>, <string-name><surname>Long</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Multimodal causal reasoning benchmark: challenging multimodal large language models to discern causal links across modalities</article-title>. In: <conf-name>Proceedings of the Findings of the Association for Computational Linguistics: ACL 2025; 2025 Jul 27&#x2013;Aug 1</conf-name>; <publisher-loc>Vienna, Austria</publisher-loc>. p. <fpage>5509</fpage>&#x2013;<lpage>33</lpage>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Ke</surname> <given-names>F</given-names></string-name>, <string-name><surname>Hsu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Cai</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Explain before you answer: a survey on compositional visual reasoning</article-title>. <comment>arXiv:2508.17298. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2508.17298</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Perception, reason, think, and plan: a survey on large multimodal reasoning models</article-title>. <comment>arXiv:2505.04921. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2505.04921</pub-id>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Lin</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>X</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Sang</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Mind with eyes: from language reasoning to multimodal reasoning</article-title>. <comment>arXiv:2503.18071. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2503.18071</pub-id>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhong</surname> <given-names>W</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>N</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A survey on large language model-powered autonomous driving</article-title>. <source>Engineering</source>. <year>2025</year>;<volume>12</volume>(<issue>1&#x2013;3</issue>):<fpage>1</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.eng.2025.07.038</pub-id>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ahn</surname> <given-names>J</given-names></string-name>, <string-name><surname>Verma</surname> <given-names>R</given-names></string-name>, <string-name><surname>Lou</surname> <given-names>R</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>D</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name>, <string-name><surname>Yin</surname> <given-names>W</given-names></string-name></person-group>. <article-title>Large language models for mathematical reasoning: progresses and challenges</article-title>. In: <conf-name>Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics: Student Research Workshop; 2024 Mar 17&#x2013;18</conf-name>; <publisher-loc>St. Julians, Malta</publisher-loc>. p. <fpage>225</fpage>&#x2013;<lpage>37</lpage>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Bai</surname> <given-names>T</given-names></string-name>, <string-name><surname>Liang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>B</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>X</given-names></string-name>, <string-name><surname>Li</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A survey of multimodal large language model from a data-centric perspective</article-title>. <comment>arXiv:2405.16640. 2024</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2405.16640</pub-id>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Luo</surname> <given-names>H</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>C</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>P</given-names></string-name>, <string-name><surname>Lou</surname> <given-names>JG</given-names></string-name>, <string-name><surname>Tao</surname> <given-names>C</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>WizardMath: empowering mathematical reasoning for large language models via reinforced evol-instruct</article-title>. In: <conf-name>Proceedings of the Thirteenth International Conference on Learning Representations; 2025 Apr 24&#x2013;28</conf-name>; <publisher-loc>Singapore, Singapore</publisher-loc>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Toshniwal</surname> <given-names>S</given-names></string-name>, <string-name><surname>Du</surname> <given-names>W</given-names></string-name>, <string-name><surname>Moshkov</surname> <given-names>I</given-names></string-name>, <string-name><surname>Kisacanin</surname> <given-names>B</given-names></string-name>, <string-name><surname>Ayrapetyan</surname> <given-names>A</given-names></string-name>, <string-name><surname>Gitman</surname> <given-names>I</given-names></string-name></person-group>. <article-title>OpenMathInstruct-2: accelerating AI for math with massive open-source instruction data</article-title>. In: <conf-name> Proceedings of the 4th Workshop on Mathematical Reasoning and AI at NeurIPS&#x2019;24; 2024 Dec 14&#x2013;15</conf-name>; <publisher-loc>Vancouver, BC, Canada</publisher-loc>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhou</surname> <given-names>G</given-names></string-name>, <string-name><surname>Qiu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Reinforced MLLM: a survey on RL-based reasoning in multimodal large language models</article-title>. <comment>arXiv:2504.21277. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2504.21277</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wei</surname> <given-names>F</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>W</given-names></string-name>, <string-name><surname>Mao</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xia</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ge</surname> <given-names>T</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>LLM as a mastermind: a survey of strategic reasoning with large language models</article-title>. <comment>arXiv:2404.01230. 2024</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2404.01230</pub-id>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bober-Irizar</surname> <given-names>M</given-names></string-name>, <string-name><surname>Banerjee</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Neural networks for abstraction and reasoning</article-title>. <source>Sci Rep</source>. <year>2024</year>;<volume>14</volume>(<issue>1</issue>):<fpage>27823</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-024-73582-7</pub-id>; <pub-id pub-id-type="pmid">39537641</pub-id></mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>F</given-names></string-name>, <string-name><surname>Jia</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>SC</given-names></string-name></person-group>. <article-title>RAVEN: a dataset for relational and analogical visual REasoNing</article-title>. In: <conf-name>Proceedings of the 2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2019 Jun 15&#x2013;20</conf-name>; <publisher-loc>Long Beach, CA, USA</publisher-loc>. p. <fpage>5312</fpage>&#x2013;<lpage>22</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr.2019.00546</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Data-and knowledge-driven visual abductive reasoning</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. <year>2026</year>;<volume>48</volume>(<issue>1</issue>):<fpage>792</fpage>&#x2013;<lpage>806</lpage>. doi:<pub-id pub-id-type="doi">10.1109/tpami.2025.3613712</pub-id>; <pub-id pub-id-type="pmid">40986577</pub-id></mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Bi</surname> <given-names>J</given-names></string-name>, <string-name><surname>Liang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>X</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>J</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Why reasoning matters? A survey of advancements in multimodal reasoning (v1)</article-title>. <comment>arXiv:2504.03151. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2504.03151</pub-id>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>W</given-names></string-name>, <string-name><surname>Han</surname> <given-names>X</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>H</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Exploring the reasoning abilities of multimodal large language models (MLLMs): a comprehensive survey on emerging trends in multimodal reasoning</article-title>. <comment>arXiv:2401.06805. 2024</comment>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yan</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Su</surname> <given-names>J</given-names></string-name>, <string-name><surname>He</surname> <given-names>J</given-names></string-name>, <string-name><surname>Fu</surname> <given-names>F</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>X</given-names></string-name>, <string-name><surname>Lyu</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A survey of mathematical reasoning in the era of multimodal large language model: benchmark, method &#x0026; challenges</article-title>. In: <conf-name>Proceedings of the Findings of the Association for Computational Linguistics: ACL 2025; 2025 Jul 27&#x2013;Aug 1</conf-name>; <publisher-loc>Vienna, Austria</publisher-loc>. p. <fpage>11798</fpage>&#x2013;<lpage>827</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2025.findings-acl.614</pub-id>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>He</surname> <given-names>C</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>R</given-names></string-name>, <string-name><surname>Bai</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Thai</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>OlympiadBench: a challenging benchmark for promoting AGI with olympiad-level bilingual multimodal scientific problems</article-title>. In: <conf-name>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers); 2024 Aug 11&#x2013;16</conf-name>; <publisher-loc>Bangkok, Thailand</publisher-loc>. p. <fpage>3828</fpage>&#x2013;<lpage>50</lpage>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>W</given-names></string-name>, <string-name><surname>Yin</surname> <given-names>M</given-names></string-name>, <string-name><surname>Ku</surname> <given-names>M</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>TheoremQA: a theorem-driven question answering dataset</article-title>. In: <conf-name>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing; 2023 Dec 6&#x2013;10</conf-name>; <publisher-loc>Singapore, Singapore</publisher-loc>. p. <fpage>7889</fpage>&#x2013;<lpage>901</lpage>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Fu</surname> <given-names>C</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>YF</given-names></string-name>, <string-name><surname>Yin</surname> <given-names>S</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Fang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>MME-survey: a comprehensive survey on evaluation of multimodal LLMs</article-title>. <comment>arXiv:2411.15296. 2024</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2411.15296</pub-id>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Ge</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ge</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>G</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>R</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>SEED-bench: benchmarking multimodal large language models</article-title>. In: <conf-name>Proceedings of the 2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2024 Jun 16&#x2013;22</conf-name>; <publisher-loc>Seattle, WA, USA</publisher-loc>. p. <fpage>13299</fpage>&#x2013;<lpage>308</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr52733.2024.01263</pub-id>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yue</surname> <given-names>X</given-names></string-name>, <string-name><surname>Ni</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>K</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>R</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>G</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>MMMU: a massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI</article-title>. In: <conf-name>Proceedings of the 2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2024 Jun 16&#x2013;22</conf-name>; <publisher-loc>Seattle, WA, USA</publisher-loc>. p. <fpage>9556</fpage>&#x2013;<lpage>67</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr52733.2024.00913</pub-id>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Multimodal chain-of-thought reasoning: a comprehensive survey</article-title>. <comment>arXiv:2503.12605. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2503.12605</pub-id>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zellers</surname> <given-names>R</given-names></string-name>, <string-name><surname>Bisk</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Farhadi</surname> <given-names>A</given-names></string-name>, <string-name><surname>Choi</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>From recognition to cognition: visual commonsense reasoning</article-title>. In: <conf-name>Proceedings of the 2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2019 Jun 15&#x2013;20</conf-name>; <publisher-loc>Long Beach, CA, USA</publisher-loc>. p. <fpage>6713</fpage>&#x2013;<lpage>24</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr.2019.00688</pub-id>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Komanduri</surname> <given-names>A</given-names></string-name>, <string-name><surname>Bhaila</surname> <given-names>K</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>X</given-names></string-name></person-group>. <article-title>CausalVLBench: benchmarking visual causal reasoning in large vision-language models</article-title>. In: <conf-name>Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing; 2025 Nov 4&#x2013;9</conf-name>; <publisher-loc>Suzhou, China</publisher-loc>. p. <fpage>30660</fpage>&#x2013;<lpage>80</lpage>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>CausalBench: a comprehensive benchmark for evaluating causal reasoning capabilities of large language models</article-title>. In: <conf-name>Proceedings of the 10th SIGHAN Workshop on Chinese Language Processing (SIGHAN-10); 2024 Aug 11&#x2013;16</conf-name>; <publisher-loc>Bangkok, Thailand</publisher-loc>. p. <fpage>143</fpage>&#x2013;<lpage>51</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2024.sighan-1.17</pub-id>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Mitra</surname> <given-names>C</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Darrell</surname> <given-names>T</given-names></string-name>, <string-name><surname>Herzig</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Compositional chain-of-thought prompting for large multimodal models</article-title>. In: <conf-name>Proceedings of the 2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2024 Jun 16&#x2013;22</conf-name>; <publisher-loc>Seattle, WA, USA</publisher-loc>. p. <fpage>14420</fpage>&#x2013;<lpage>31</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr52733.2024.01367</pub-id>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>OpenAI, Achiam</surname> <given-names>J</given-names></string-name>, <string-name><surname>Adler</surname> <given-names>S</given-names></string-name>, <string-name><surname>Agarwal</surname> <given-names>S</given-names></string-name>, <string-name><surname>Ahmad</surname> <given-names>L</given-names></string-name>, <string-name><surname>Akkaya</surname> <given-names>I</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>GPT-4 technical report</article-title>. <comment>arXiv:2303.08774. 2024</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2303.08774</pub-id>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Besta</surname> <given-names>M</given-names></string-name>, <string-name><surname>Barth</surname> <given-names>J</given-names></string-name>, <string-name><surname>Schreiber</surname> <given-names>E</given-names></string-name>, <string-name><surname>Kubicek</surname> <given-names>A</given-names></string-name>, <string-name><surname>Catarino</surname> <given-names>A</given-names></string-name>, <string-name><surname>Gerstenberger</surname> <given-names>R</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Reasoning language models: a blueprint</article-title>. <comment>arXiv:2501.11223. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2501.11223</pub-id>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Johnson</surname> <given-names>J</given-names></string-name>, <string-name><surname>Hariharan</surname> <given-names>B</given-names></string-name>, <string-name><surname>van der Maaten</surname> <given-names>L</given-names></string-name>, <string-name><surname>Li</surname> <given-names>FF</given-names></string-name>, <string-name><surname>Zitnick</surname> <given-names>CL</given-names></string-name>, <string-name><surname>Girshick</surname> <given-names>R</given-names></string-name></person-group>. <article-title>CLEVR: a diagnostic dataset for compositional language and elementary visual reasoning</article-title>. In: <conf-name>Proceedings of the 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR); 2017 Jul 21&#x2013;26</conf-name>; <publisher-loc>Honolulu, HI, USA</publisher-loc>. p. <fpage>1988</fpage>&#x2013;<lpage>97</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr.2017.215</pub-id>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Li</surname> <given-names>D</given-names></string-name>, <string-name><surname>Savarese</surname> <given-names>S</given-names></string-name>, <string-name><surname>Hoi</surname> <given-names>S</given-names></string-name></person-group>. <article-title>BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models</article-title>. <comment>arXiv:2301.12597v3. 2023</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2301.12597</pub-id>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yi</surname> <given-names>K</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Gan</surname> <given-names>C</given-names></string-name>, <string-name><surname>Torralba</surname> <given-names>A</given-names></string-name>, <string-name><surname>Kohli</surname> <given-names>P</given-names></string-name>, <string-name><surname>Tenenbaum</surname> <given-names>JB</given-names></string-name></person-group>. <article-title>Neural-symbolic VQA: disentangling reasoning from vision and language understanding</article-title>. In: <conf-name>Proceedings of the 32nd International Conference on Neural Information Processing Systems. NIPS&#x2019;18; 2018 Dec 3&#x2013;8; Montr&#x00E9;al, QC, Canada</conf-name>. <publisher-loc>Red Hook, NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc.</publisher-name>; <year>2018</year>. p. <fpage>1039</fpage>&#x2013;<lpage>50</lpage>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Gou</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Shao</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Gong</surname> <given-names>Y</given-names></string-name>, <string-name><surname>yelong</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>M</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>ToRA: a tool-integrated reasoning agent for mathematical problem solving</article-title>. In: <conf-name>Proceedings of the Twelfth International Conference on Learning Representations; 2024 May 7&#x2013;11</conf-name>; <publisher-loc>Vienna, Austria</publisher-loc>.</mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>W</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Cohen</surname> <given-names>WW</given-names></string-name></person-group>. <article-title>Program of thoughts prompting: disentangling computation from reasoning for numerical reasoning tasks</article-title>. <year>2023 [cited 2026 Jan 1]</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=YfZ4ZPt8zd">https://openreview.net/forum?id&#x003D;YfZ4ZPt8zd</ext-link>.</mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>J</given-names></string-name>, <string-name><surname>Schuurmans</surname> <given-names>D</given-names></string-name>, <string-name><surname>Le</surname> <given-names>QV</given-names></string-name>, <string-name><surname>Chi</surname> <given-names>EH</given-names></string-name>, <string-name><surname>Narang</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Self-consistency improves chain of thought reasoning in language models</article-title>. In: <conf-name>Proceedings of the Eleventh International Conference on Learning Representations; 2023 May 1&#x2013;5</conf-name>; <publisher-loc>Kigali, Rwanda</publisher-loc>.</mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Duan</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>W</given-names></string-name>, <etal>et al</etal></person-group>. <chapter-title>MMBench: is your multi-modal model an all-around player?</chapter-title> In: <article-title>Proceedings of the Computer Vision&#x2014;ECCV 2024: 18th European Conference; 2024 Sep 29&#x2013;Oct 4</article-title>; <publisher-loc>Milan, Italy</publisher-loc>. p. <fpage>216</fpage>&#x2013;<lpage>33</lpage>.</mixed-citation></ref>
<ref id="ref-46"><label>[46]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Cai</surname> <given-names>H</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>W</given-names></string-name></person-group>. <article-title>MM-IQ: benchmarking human-like abstraction and reasoning in multimodal models</article-title>. <comment>arXiv:2502.00698. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2502.00698</pub-id>.</mixed-citation></ref>
<ref id="ref-47"><label>[47]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>An early evaluation of GPT-4V(ision)</article-title>. <comment>arXiv:2310.16534. 2023</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2310.16534</pub-id>.</mixed-citation></ref>
<ref id="ref-48"><label>[48]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>A</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>J</given-names></string-name>, <string-name><surname>Qin</surname> <given-names>L</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>GPT-4V(ision) as a generalist evaluator for vision-language tasks</article-title>. <comment>arXiv:2311.01361. 2023</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2311.01361</pub-id>.</mixed-citation></ref>
<ref id="ref-49"><label>[49]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>W</given-names></string-name>, <string-name><surname>Vergari</surname> <given-names>A</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Can VLMs reason robustly? A neuro-symbolic investigation</article-title>. <comment>arXiv:2603.23867. 2026</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2603.23867</pub-id>.</mixed-citation></ref>
<ref id="ref-50"><label>[50]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Dong</surname> <given-names>J</given-names></string-name>, <string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Su</surname> <given-names>D</given-names></string-name>, <string-name><surname>Chu</surname> <given-names>C</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>MM-LLMs: recent advances in MultiModal large language models</article-title>. In: <conf-name>Proceedings of the Findings of the Association for Computational Linguistics ACL 2024; 2024 Aug 11&#x2013;16</conf-name>; <publisher-loc>Bangkok, Thailand</publisher-loc>. p. <fpage>12401</fpage>&#x2013;<lpage>30</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2024.findings-acl.738</pub-id>.</mixed-citation></ref>
<ref id="ref-51"><label>[51]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Colelough</surname> <given-names>BC</given-names></string-name>, <string-name><surname>Regli</surname> <given-names>W</given-names></string-name></person-group>. <article-title>Neuro-symbolic AI in 2024: a systematic review</article-title>. <comment>arXiv:2501.05435. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2501.05435</pub-id>.</mixed-citation></ref>
<ref id="ref-52"><label>[52]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Radford</surname> <given-names>A</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>JW</given-names></string-name>, <string-name><surname>Hallacy</surname> <given-names>C</given-names></string-name>, <string-name><surname>Ramesh</surname> <given-names>A</given-names></string-name>, <string-name><surname>Goh</surname> <given-names>G</given-names></string-name>, <string-name><surname>Agarwal</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Learning transferable visual models from natural language supervision</article-title>. In: <conf-name>Proceedings of the 38th International Conference on Machine Learning; 2021 Jul 18&#x2013;24</conf-name>; <publisher-loc>Virtual</publisher-loc>. p. <fpage>8748</fpage>&#x2013;<lpage>63</lpage>.</mixed-citation></ref>
<ref id="ref-53"><label>[53]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>YJ</given-names></string-name></person-group>. <article-title>Improved baselines with visual instruction tuning</article-title>. In: <conf-name>Proceedings of the 2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2024 Jun 16&#x2013;22</conf-name>; <publisher-loc>Seattle, WA, USA</publisher-loc>. p. <fpage>26286</fpage>&#x2013;<lpage>96</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr52733.2024.02484</pub-id>.</mixed-citation></ref>
<ref id="ref-54"><label>[54]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Peng</surname> <given-names>B</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Bo</surname> <given-names>X</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Fan</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>M3GQA: a multi-entity multi-hop multi-setting graph question answering benchmark</article-title>. In: <conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers); 2025 Jul 27&#x2013;Aug 1</conf-name>; <publisher-loc>Vienna, Austria</publisher-loc>. p. <fpage>30594</fpage>&#x2013;<lpage>620</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.1478</pub-id>.</mixed-citation></ref>
<ref id="ref-55"><label>[55]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Xiang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Han</surname> <given-names>SC</given-names></string-name>, <string-name><surname>Ding</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>BRIDGE: benchmark for multi-hop reasoning in long multimodal documents with grounded evidence</article-title>. <comment>arXiv:2603.07931. 2026</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2603.07931</pub-id>.</mixed-citation></ref>
<ref id="ref-56"><label>[56]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Tang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>MultiHop-RAG: benchmarking retrieval-augmented generation for multi-hop queries</article-title>. In: <conf-name>Proceedings of the First Conference on Language Modeling; 2024 Oct 7&#x2013;9</conf-name>; <publisher-loc>Philadelphia, PA, USA</publisher-loc>.</mixed-citation></ref>
<ref id="ref-57"><label>[57]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yao</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>D</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>J</given-names></string-name>, <string-name><surname>Shafran</surname> <given-names>I</given-names></string-name>, <string-name><surname>Griffiths</surname> <given-names>T</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Tree of thoughts: deliberate problem solving with large language models</article-title>. <source>Adv Neural Inf Process Syst</source>. <year>2023</year>;<volume>36</volume>:<fpage>11809</fpage>&#x2013;<lpage>22</lpage>.</mixed-citation></ref>
<ref id="ref-58"><label>[58]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Shi</surname> <given-names>W</given-names></string-name>, <string-name><surname>Fu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Roth</surname> <given-names>D</given-names></string-name>, <string-name><surname>Ostendorf</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zettlemoyer</surname> <given-names>L</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Visual SKETCHPAD: sketching as a visual chain of thought for multimodal language models</article-title>. In: <conf-name>Proceedings of the 38th International Conference on Neural Information Processing Systems NIPS &#x2019;24; 2024 Dec 10&#x2013;15</conf-name>; <publisher-loc>Vancouver, BC, Canada</publisher-loc>.</mixed-citation></ref>
<ref id="ref-59"><label>[59]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>R</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>X</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Kang</surname> <given-names>Z</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>DHCP: detecting hallucinations by cross-modal attention pattern in large vision-language models</article-title>. <comment>arXiv:2411.18659. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2411.18659</pub-id>.</mixed-citation></ref>
<ref id="ref-60"><label>[60]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ye</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Feng</surname> <given-names>X</given-names></string-name>, <string-name><surname>Qin</surname> <given-names>L</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>CLAIM: mitigating multilingual object hallucination in large vision-language models with cross-lingual attention intervention</article-title>. In: <conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers); 2025 Jul 27&#x2013;Aug 1</conf-name>; <publisher-loc>Vienna, Austria</publisher-loc>. p. <fpage>13080</fpage>&#x2013;<lpage>94</lpage>.</mixed-citation></ref>
<ref id="ref-61"><label>[61]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Changpinyo</surname> <given-names>S</given-names></string-name>, <string-name><surname>Piergiovanni</surname> <given-names>A</given-names></string-name>, <string-name><surname>Padlewski</surname> <given-names>P</given-names></string-name>, <string-name><surname>Salz</surname> <given-names>D</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>PaLI: a jointly-scaled multilingual language-image model</article-title>. In: <conf-name>Proceedings of the Eleventh International Conference on Learning Representations; 2023 May 1&#x2013;5</conf-name>; <publisher-loc>Kigali, Rwanda</publisher-loc>.</mixed-citation></ref>
<ref id="ref-62"><label>[62]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Olausson</surname> <given-names>T</given-names></string-name>, <string-name><surname>Gu</surname> <given-names>A</given-names></string-name>, <string-name><surname>Lipkin</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Solar-Lezama</surname> <given-names>A</given-names></string-name>, <string-name><surname>Tenenbaum</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>LINC: a neurosymbolic approach for logical reasoning by combining language models with first-order logic provers</article-title>. In: <conf-name>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing; 2023 Dec 6&#x2013;10</conf-name>; <publisher-loc>Singapore, Singapore</publisher-loc>. p. <fpage>5153</fpage>&#x2013;<lpage>76</lpage>.</mixed-citation></ref>
<ref id="ref-63"><label>[63]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Mao</surname> <given-names>J</given-names></string-name>, <string-name><surname>Gan</surname> <given-names>C</given-names></string-name>, <string-name><surname>Kohli</surname> <given-names>P</given-names></string-name>, <string-name><surname>Tenenbaum</surname> <given-names>JB</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>J</given-names></string-name></person-group>. <article-title>The neuro-symbolic concept learner: interpreting scenes, words, and sentences from natural supervision</article-title>. In: <conf-name>International Conference on Learning Representations</conf-name>; <year>2019</year> <comment>May 6&#x2013;9; New Orleans, LA, USA</comment>.</mixed-citation></ref>
<ref id="ref-64"><label>[64]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yu</surname> <given-names>W</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Dong</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Feng</surname> <given-names>J</given-names></string-name></person-group>. <article-title>ReClor: a reading comprehension dataset requiring logical reasoning</article-title>. In: <conf-name>International Conference on Learning Representations</conf-name>; <year>2020</year> <comment>Apr 26&#x2013;30; Addis Ababa, Ethiopia</comment>.</mixed-citation></ref>
<ref id="ref-65"><label>[65]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>He</surname> <given-names>X</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>H</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Deng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>R1-onevision: advancing generalized multimodal reasoning through cross-modal formalization</article-title>. In: <conf-name>Proceedings of the 2025 IEEE/CVF International Conference on Computer Vision (ICCV); 2025 Oct 19&#x2013;25</conf-name>; <publisher-loc>Honolulu, HI, USA</publisher-loc>. p. <fpage>2376</fpage>&#x2013;<lpage>85</lpage>. doi:<pub-id pub-id-type="doi">10.1109/iccv51701.2025.00229</pub-id>.</mixed-citation></ref>
<ref id="ref-66"><label>[66]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>LT</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>R</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Towards multimodal inductive learning: adaptively embedding MMKG via prototypes</article-title>. In: <conf-name>Proceedings of the ACM on Web Conference 2025; 2025 Apr 28&#x2013;May 2</conf-name>; <publisher-loc>Sydney, Australia</publisher-loc>. p. <fpage>109</fpage>&#x2013;<lpage>18</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3696410.3714781</pub-id>.</mixed-citation></ref>
<ref id="ref-67"><label>[67]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Sreekumar</surname> <given-names>G</given-names></string-name>, <string-name><surname>Boddeti</surname> <given-names>VN</given-names></string-name></person-group>. <article-title>InPhyRe discovers: large multimodal models struggle in inductive physical reasoning</article-title>. <comment>arXiv:2509.12263. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2509.12263</pub-id>.</mixed-citation></ref>
<ref id="ref-68"><label>[68]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>F</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Yin</surname> <given-names>D</given-names></string-name>, <string-name><surname>Rong</surname> <given-names>K</given-names></string-name>, <string-name><surname>Rao</surname> <given-names>F</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>WeThink: toward general-purpose vision-language reasoning via reinforcement learning</article-title>. <comment>arXiv:2506.07905. 2025</comment>.</mixed-citation></ref>
<ref id="ref-69"><label>[69]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Deitke</surname> <given-names>M</given-names></string-name>, <string-name><surname>Clark</surname> <given-names>C</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>S</given-names></string-name>, <string-name><surname>Tripathi</surname> <given-names>R</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Park</surname> <given-names>JS</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Molmo and PixMo: open weights and open data for state-of-the-art vision-language models</article-title>. In: <conf-name>Proceedings of the 2025 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2025 Jun 10&#x2013;17</conf-name>; <publisher-loc>Nashville, TN, USA</publisher-loc>. p. <fpage>91</fpage>&#x2013;<lpage>104</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr52734.2025.00018</pub-id>.</mixed-citation></ref>
<ref id="ref-70"><label>[70]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Huang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Jia</surname> <given-names>B</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>S</given-names></string-name>, <string-name><surname>Ye</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>F</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Z</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Vision-R1: incentivizing reasoning capability in multimodal large language models</article-title>. In: <conf-name>Proceedings of the Fourteenth International Conference on Learning Representations; 2026 Apr 23&#x2013;27</conf-name>; <publisher-loc>Rio de Janeiro, Brazil</publisher-loc>.</mixed-citation></ref>
<ref id="ref-71"><label>[71]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chang</surname> <given-names>KW</given-names></string-name>, <string-name><surname>Cheng</surname> <given-names>H</given-names></string-name>, <string-name><surname>Galley</surname> <given-names>M</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Peng</surname> <given-names>B</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Chameleon: plug-and-play compositional reasoning with large language models</article-title>. <source>Adv Neural Inf Process Syst</source>. <year>2023</year>;<volume>36</volume>:<fpage>43447</fpage>&#x2013;<lpage>78</lpage>.</mixed-citation></ref>
<ref id="ref-72"><label>[72]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>F</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Li</surname> <given-names>W</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>LLaVA-NeXT-interleave: tackling multi-image, video, and 3D in large multimodal models</article-title>. <comment>arXiv:2407.07895. 2024</comment>.</mixed-citation></ref>
<ref id="ref-73"><label>[73]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Bai</surname> <given-names>S</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>K</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Ge</surname> <given-names>W</given-names></string-name>, <string-name><surname>Song</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Qwen2.5-VL technical report</article-title>. <comment>arXiv:2502.13923. 2025</comment>.</mixed-citation></ref>
<ref id="ref-74"><label>[74]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Ju</surname> <given-names>X</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Mao</surname> <given-names>J</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>EmbodiedVSR: dynamic scene graph-guided chain-of-thought reasoning for visual spatial tasks</article-title>. <comment>arXiv:2503.11089. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2503.11089</pub-id>.</mixed-citation></ref>
<ref id="ref-75"><label>[75]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Mo</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zong</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>B</given-names></string-name>, <string-name><surname>Yim</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>DixitWorld: evaluating multimodal abductive reasoning in vision-language models with multi-agent dixit gameplay</article-title>. <comment>arXiv:2510.10117. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2510.10117</pub-id>.</mixed-citation></ref>
<ref id="ref-76"><label>[76]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tan</surname> <given-names>K</given-names></string-name>, <string-name><surname>Qi</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhong</surname> <given-names>J</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>W</given-names></string-name></person-group>. <article-title>KN-VLM: KNowledge-guided vision-and-language model for visual abductive reasoning</article-title>. <source>Multimed Syst</source>. <year>2025</year>;<volume>31</volume>(<issue>2</issue>):<fpage>146</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s00530-025-01683-y</pub-id>.</mixed-citation></ref>
<ref id="ref-77"><label>[77]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhu</surname> <given-names>W</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>A</given-names></string-name>, <string-name><surname>Song</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>K</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>C</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Z</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>From perception to reasoning: deep thinking empowers multimodal large language models</article-title>. <comment>arXiv:2511.12861. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2511.12861</pub-id>.</mixed-citation></ref>
<ref id="ref-78"><label>[78]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Lu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Bansal</surname> <given-names>H</given-names></string-name>, <string-name><surname>Xia</surname> <given-names>T</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Hajishirzi</surname> <given-names>H</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>MathVista: evaluating mathematical reasoning of foundation models in visual contexts</article-title>. In: <conf-name>Proceedings of the Twelfth International Conference on Learning Representations; 2024 May 7&#x2013;11</conf-name>; <publisher-loc>Vienna, Austria</publisher-loc>.</mixed-citation></ref>
<ref id="ref-79"><label>[79]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yue</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>G</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>W</given-names></string-name></person-group>. <article-title>MAmmoTH2: scaling instructions from the web</article-title>. In: <conf-name>Proceedings of the 38th International Conference on Neural Information Processing Systems NIPS &#x2019;24; 2024 Dec 10&#x2013;15; Vancouver, BC, Canada</conf-name>.</mixed-citation></ref>
<ref id="ref-80"><label>[80]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>R</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>H</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Qiu</surname> <given-names>P</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Mathverse: does your multi-modal LLM truly see the diagrams in visual math problems?</article-title> In: <conf-name>Computer Vision&#x2014;ECCV 2024</conf-name>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name>; <year>2025</year>. p. <fpage>169</fpage>&#x2013;<lpage>86</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-031-73242-3_10</pub-id>.</mixed-citation></ref>
<ref id="ref-81"><label>[81]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Qin</surname> <given-names>J</given-names></string-name>, <string-name><surname>Liang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Xing</surname> <given-names>E</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>GeoQA: a geometric question answering benchmark towards multimodal numerical reasoning</article-title>. In: <conf-name>Proceedings of the Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021; 2021 Aug 1&#x2013;6</conf-name>; <publisher-loc>Virtual</publisher-loc>. p. <fpage>513</fpage>&#x2013;<lpage>23</lpage>.</mixed-citation></ref>
<ref id="ref-82"><label>[82]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Marino</surname> <given-names>K</given-names></string-name>, <string-name><surname>Rastegari</surname> <given-names>M</given-names></string-name>, <string-name><surname>Farhadi</surname> <given-names>A</given-names></string-name>, <string-name><surname>Mottaghi</surname> <given-names>R</given-names></string-name></person-group>. <article-title>OK-VQA: a visual question answering benchmark requiring external knowledge</article-title>. In: <conf-name>Proceedings of the 2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2019 Jun 15&#x2013;20</conf-name>; <publisher-loc>Long Beach, CA, USA</publisher-loc>. p. <fpage>3190</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr.2019.00331</pub-id>.</mixed-citation></ref>
<ref id="ref-83"><label>[83]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Lei</surname> <given-names>J</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Bansal</surname> <given-names>M</given-names></string-name>, <string-name><surname>Berg</surname> <given-names>T</given-names></string-name></person-group>. <article-title>TVQA: localized, compositional video question answering</article-title>. In: <conf-name>Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing; 2018 Oct 31&#x2013;Nov 4</conf-name>; <publisher-loc>Brussels, Belgium</publisher-loc>. p. <fpage>1369</fpage>&#x2013;<lpage>79</lpage>.</mixed-citation></ref>
<ref id="ref-84"><label>[84]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>D</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhuang</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>ActivityNet-QA: a dataset for understanding complex web videos via question answering</article-title>. In: <conf-name>Proceedings of the Thirty-Third AAAI Conference on Artificial Intelligence and Thirty-First Innovative Applications of Artificial Intelligence Conference and Ninth AAAI Symposium on Educational Advances in Artificial Intelligence AAAI&#x2019;19/IAAI&#x2019;19/EAAI&#x2019;19; 2019 Jan 27&#x2013;Feb 1</conf-name>; <publisher-loc>Honolulu, HI, USA</publisher-loc>.</mixed-citation></ref>
<ref id="ref-85"><label>[85]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Li</surname> <given-names>X</given-names></string-name>, <string-name><surname>Bing</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Video-LLaMA: an instruction-tuned audio-visual language model for video understanding</article-title>. In: <conf-name>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing: System Demonstrations; 2023 Dec 6&#x2013;10</conf-name>; <publisher-loc>Singapore, Singapore</publisher-loc>. p. <fpage>543</fpage>&#x2013;<lpage>53</lpage>.</mixed-citation></ref>
<ref id="ref-86"><label>[86]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Abu-El-Haija</surname> <given-names>S</given-names></string-name>, <string-name><surname>Kothari</surname> <given-names>N</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>J</given-names></string-name>, <string-name><surname>Natsev</surname> <given-names>P</given-names></string-name>, <string-name><surname>Toderici</surname> <given-names>G</given-names></string-name>, <string-name><surname>Varadarajan</surname> <given-names>B</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>YouTube-8M: a large-scale video classification benchmark</article-title>. <comment>arXiv:1609.08675. 2016</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.1609.08675</pub-id>.</mixed-citation></ref>
<ref id="ref-87"><label>[87]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Stogiannidis</surname> <given-names>I</given-names></string-name>, <string-name><surname>McDonagh</surname> <given-names>S</given-names></string-name>, <string-name><surname>Tsaftaris</surname> <given-names>SA</given-names></string-name></person-group>. <article-title>Mind the gap: benchmarking spatial reasoning in vision-language models</article-title>. <comment>arXiv:2503.19707. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2503.19707</pub-id>.</mixed-citation></ref>
<ref id="ref-88"><label>[88]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Cheng</surname> <given-names>D</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>S</given-names></string-name>, <string-name><surname>Dai</surname> <given-names>J</given-names></string-name></person-group>. <article-title>An empirical study of spatial attention mechanisms in deep networks</article-title>. In: <conf-name>Proceedings of the 2019 IEEE/CVF International Conference on Computer Vision (ICCV); 2019 Oct 27&#x2013;Nov 2; Seoul, Republic of Korea</conf-name>. p. <fpage>6687</fpage>&#x2013;<lpage>96</lpage>. doi:<pub-id pub-id-type="doi">10.1109/iccv.2019.00679</pub-id>.</mixed-citation></ref>
<ref id="ref-89"><label>[89]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>K</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zeng</surname> <given-names>W</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>F</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Shikra: unleashing multimodal LLM&#x2019;s referential dialogue magic</article-title>. <comment>arXiv:2306.15195. 2023</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2306.15195</pub-id>.</mixed-citation></ref>
<ref id="ref-90"><label>[90]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Tian</surname> <given-names>W</given-names></string-name>, <string-name><surname>Jiao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Qian</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>N</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>B</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Look before you decide: prompting active deduction of MLLMs for assumptive reasoning</article-title>. In: <conf-name> Proceedings of the 33rd ACM International Conference on Multimedia; 2025 Oct 27&#x2013;31</conf-name>; <publisher-loc>Dublin, Ireland</publisher-loc>. p. <fpage>2713</fpage>&#x2013;<lpage>22</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3746027.3754720</pub-id>.</mixed-citation></ref>
<ref id="ref-91"><label>[91]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhou</surname> <given-names>S</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>FF</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>X</given-names></string-name>, <string-name><surname>Lo</surname> <given-names>R</given-names></string-name>, <string-name><surname>Sridhar</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>WebArena: a realistic web environment for building autonomous agents</article-title>. <comment>arXiv:2307.13854. 2024</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2307.13854</pub-id>.</mixed-citation></ref>
<ref id="ref-92"><label>[92]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Team</surname> <given-names>G</given-names></string-name>, <string-name><surname>Anil</surname> <given-names>R</given-names></string-name>, <string-name><surname>Borgeaud</surname> <given-names>S</given-names></string-name>, <string-name><surname>Alayrac</surname> <given-names>JB</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Soricut</surname> <given-names>R</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Gemini: a family of highly capable multimodal models</article-title>. <comment>arXiv:2312.11805. 2023</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2312.11805</pub-id>.</mixed-citation></ref>
<ref id="ref-93"><label>[93]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Alayrac</surname> <given-names>JB</given-names></string-name>, <string-name><surname>Barr</surname> <given-names>I</given-names></string-name>, <string-name><surname>Barreira</surname> <given-names>R</given-names></string-name>, <string-name><surname>Binkowski</surname> <given-names>M</given-names></string-name>, <string-name><surname>Borgeaud</surname> <given-names>S</given-names></string-name>, <string-name><surname>Brock</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Flamingo: a visual language model for few-shot learning</article-title>. <source>Adv Neural Inf Process Syst</source>. <year>2022</year>;<volume>35</volume>:<fpage>23716</fpage>&#x2013;<lpage>36</lpage>. doi:<pub-id pub-id-type="doi">10.52202/068431-1723</pub-id>.</mixed-citation></ref>
<ref id="ref-94"><label>[94]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dai</surname> <given-names>W</given-names></string-name>, <string-name><surname>Fung</surname> <given-names>PN</given-names></string-name>, <string-name><surname>Hoi</surname> <given-names>S</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Li</surname> <given-names>D</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>InstructBLIP: towards general-purpose vision-language models with instruction tuning</article-title>. <source>Adv Neural Inf Process Syst</source>. <year>2023</year>;<volume>36</volume>:<fpage>49250</fpage>&#x2013;<lpage>67</lpage>. doi:<pub-id pub-id-type="doi">10.52202/075280-2142</pub-id>.</mixed-citation></ref>
<ref id="ref-95"><label>[95]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Su</surname> <given-names>W</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>G</given-names></string-name>, <string-name><surname>Xing</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Intern VL: scaling up vision foundation models and aligning for generic visual-linguistic tasks</article-title>. In: <conf-name> Proceedings of the 2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2024 Jun 16&#x2013;22</conf-name>; <publisher-loc>Seattle, WA, USA</publisher-loc>. p. <fpage>24185</fpage>&#x2013;<lpage>98</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr52733.2024.02283</pub-id>.</mixed-citation></ref>
<ref id="ref-96"><label>[96]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhu</surname> <given-names>D</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>X</given-names></string-name>, <string-name><surname>Li</surname> <given-names>X</given-names></string-name>, <string-name><surname>Elhoseiny</surname> <given-names>M</given-names></string-name></person-group>. <article-title>MiniGPT-4: enhancing vision-language understanding with advanced large language models</article-title>. In: <conf-name>Proceedings of the Twelfth International Conference on Learning Representations; 2024 May 7&#x2013;11</conf-name>; <publisher-loc>Vienna, Austria</publisher-loc>.</mixed-citation></ref>
<ref id="ref-97"><label>[97]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Pu</surname> <given-names>F</given-names></string-name>, <string-name><surname>Cahyono</surname> <given-names>JA</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Otter: a multi-modal model with in-context instruction tuning</article-title>. <comment>arXiv:2305.03726. 2023</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2305.03726</pub-id>.</mixed-citation></ref>
<ref id="ref-98"><label>[98]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Peng</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Dong</surname> <given-names>L</given-names></string-name>, <string-name><surname>Hao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Grounding multimodal large language models to the world</article-title>. In: <conf-name>Proceedings of the Twelfth International Conference on Learning Representations; 2024 May 7&#x2013;11</conf-name>; <publisher-loc>Vienna, Austria</publisher-loc>.</mixed-citation></ref>
<ref id="ref-99"><label>[99]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>P</given-names></string-name>, <string-name><surname>Bai</surname> <given-names>S</given-names></string-name>, <string-name><surname>Tan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Fan</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Bai</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Qwen2-VL: enhancing vision-language model&#x2019;s perception of the world at any resolution</article-title>. <comment>arXiv:2409.12191. 2024</comment>.</mixed-citation></ref>
<ref id="ref-100"><label>[100]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>D</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name>, <string-name><surname>Li</surname> <given-names>F</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>LLaVA-OneVision: easy visual task transfer</article-title>. <comment>arXiv:2408.03326. 2024</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2408.03326</pub-id>.</mixed-citation></ref>
<ref id="ref-101"><label>[101]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Cui</surname> <given-names>E</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Expanding performance boundaries of open-source multimodal models with model, data, and test-time scaling</article-title>. <comment>arXiv:2412.05271. 2024</comment>.</mixed-citation></ref>
<ref id="ref-102"><label>[102]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Team</surname> <given-names>GLM-V</given-names></string-name>, <string-name><surname>Hong</surname> <given-names>W</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>W</given-names></string-name>, <string-name><surname>Gu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>G</given-names></string-name>, <string-name><surname>Gan</surname> <given-names>G</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>GLM-4.5V and GLM-4.1V-thinking: towards versatile multimodal reasoning with scalable reinforcement learning</article-title>. <comment>arXiv:2507.01006. 2025</comment>.</mixed-citation></ref>
<ref id="ref-103"><label>[103]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Tian</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>K</given-names></string-name>, <string-name><surname>Tong</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>MMaDA: multimodal large diffusion language models</article-title>. <comment>arXiv:2505.15809. 2025</comment>.</mixed-citation></ref>
<ref id="ref-104"><label>[104]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Jiang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Heng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ye</surname> <given-names>W</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>VLM-R3: region recognition, reasoning, and refinement for enhanced multimodal chain-of-thought</article-title>. In: <conf-name>Proceedings of the Thirty-Ninth Annual Conference on Neural Information Processing Systems; 2025 Nov 30&#x2013;Dec 7</conf-name>; <publisher-loc>San Diego, CA, USA</publisher-loc>.</mixed-citation></ref>
<ref id="ref-105"><label>[105]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Gao</surname> <given-names>L</given-names></string-name>, <string-name><surname>Madaan</surname> <given-names>A</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>S</given-names></string-name>, <string-name><surname>Alon</surname> <given-names>U</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>PAL: program-aided language models</article-title>. <comment>arXiv:2211.10435. 2023</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2211.10435</pub-id>.</mixed-citation></ref>
<ref id="ref-106"><label>[106]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Schick</surname> <given-names>T</given-names></string-name>, <string-name><surname>Dwivedi-Yu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Dessi</surname> <given-names>R</given-names></string-name>, <string-name><surname>Raileanu</surname> <given-names>R</given-names></string-name>, <string-name><surname>Lomeli</surname> <given-names>M</given-names></string-name>, <string-name><surname>Hambro</surname> <given-names>E</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Toolformer: language models can teach themselves to use tools</article-title>. <source>Adv Neural Inf Process Syst</source>. <year>2023</year>;<volume>36</volume>:<fpage>68539</fpage>&#x2013;<lpage>51</lpage>.</mixed-citation></ref>
<ref id="ref-107"><label>[107]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Gupta</surname> <given-names>T</given-names></string-name>, <string-name><surname>Kembhavi</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Visual programming: compositional visual reasoning without training</article-title>. <comment>arXiv:2211.11559. 2022</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2211.11559</pub-id>.</mixed-citation></ref>
<ref id="ref-108"><label>[108]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Simoulin</surname> <given-names>A</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Code to think, think to code: a survey on code-enhanced reasoning and reasoning-driven code intelligence in LLMs</article-title>. In: <conf-name>Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing; 2025 Nov 4&#x2013;9</conf-name>; <publisher-loc>Suzhou, China</publisher-loc>. p. <fpage>2586</fpage>&#x2013;<lpage>616</lpage>.</mixed-citation></ref>
<ref id="ref-109"><label>[109]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Jia</surname> <given-names>C</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xia</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>YT</given-names></string-name>, <string-name><surname>Parekh</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Pham</surname> <given-names>H</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Scaling up visual and vision-language representation learning with noisy text supervision</article-title>. In: <conf-name>Proceedings of the 38th International Conference on Machine Learning; 2021 Jul 18&#x2013;24</conf-name>; <publisher-loc>Virtual</publisher-loc>. p. <fpage>4904</fpage>&#x2013;<lpage>16</lpage>.</mixed-citation></ref>
<ref id="ref-110"><label>[110]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Bao</surname> <given-names>H</given-names></string-name>, <string-name><surname>Dong</surname> <given-names>L</given-names></string-name>, <string-name><surname>Piao</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>F</given-names></string-name></person-group>. <article-title>BEiT: BERT pre-training of image transformers</article-title>. <comment>arXiv:2106.08254. 2022</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2106.08254</pub-id>.</mixed-citation></ref>
<ref id="ref-111"><label>[111]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Krishna</surname> <given-names>R</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Groth</surname> <given-names>O</given-names></string-name>, <string-name><surname>Johnson</surname> <given-names>J</given-names></string-name>, <string-name><surname>Hata</surname> <given-names>K</given-names></string-name>, <string-name><surname>Kravitz</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Visual genome: connecting language and vision using crowdsourced dense image annotations</article-title>. <comment>arXiv:1602.07332. 2016</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.1602.07332</pub-id>.</mixed-citation></ref>
<ref id="ref-112"><label>[112]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Yu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Li</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhuang</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Deep modular co-attention networks for visual question answering</article-title>. <comment>arXiv:1906.10770. 2019</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.1906.10770</pub-id>.</mixed-citation></ref>
<ref id="ref-113"><label>[113]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Caesar</surname> <given-names>H</given-names></string-name>, <string-name><surname>Uijlings</surname> <given-names>J</given-names></string-name>, <string-name><surname>Ferrari</surname> <given-names>V</given-names></string-name></person-group>. <article-title>COCO-stuff: thing and stuff classes in context</article-title>. <comment>arXiv:1612.03716. 2018</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.1612.03716</pub-id>.</mixed-citation></ref>
<ref id="ref-114"><label>[114]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Bao</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Dong</surname> <given-names>L</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Mohammed</surname> <given-names>OK</given-names></string-name>, <string-name><surname>Aggarwal</surname> <given-names>K</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>VLMo: unified vision-language pre-training with mixture-of-modality-experts</article-title>. <comment>arXiv:2111.02358. 2022</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2111.02358</pub-id>.</mixed-citation></ref>
<ref id="ref-115"><label>[115]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Jia</surname> <given-names>L</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Think3D: thinking with space for spatial reasoning</article-title>. <comment>arXiv:2601.13029. 2026</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2601.13029</pub-id>.</mixed-citation></ref>
<ref id="ref-116"><label>[116]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>A</given-names></string-name>, <string-name><surname>Li</surname> <given-names>M</given-names></string-name>, <string-name><surname>hai</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Karypis</surname> <given-names>G</given-names></string-name>, <string-name><surname>Smola</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Multimodal chain-of-thought reasoning in language models</article-title>. <year>2024 [cited 2026 Jan 1]</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=y1pPWFVfvR">https://openreview.net/forum?id&#x003D;y1pPWFVfvR</ext-link>.</mixed-citation></ref>
<ref id="ref-117"><label>[117]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Min</surname> <given-names>S</given-names></string-name>, <string-name><surname>Deng</surname> <given-names>X</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zettlemoyer</surname> <given-names>L</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Towards understanding chain-of-thought prompting: an empirical study of what matters</article-title>. <comment>arXiv:2212.10001. 2023</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2212.10001</pub-id>.</mixed-citation></ref>
<ref id="ref-118"><label>[118]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>A</given-names></string-name>, <string-name><surname>Li</surname> <given-names>M</given-names></string-name>, <string-name><surname>Smola</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Automatic chain of thought prompting in large language models</article-title>. <comment>arXiv:2210.03493. 2022</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2210.03493</pub-id>.</mixed-citation></ref>
<ref id="ref-119"><label>[119]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Rose</surname> <given-names>D</given-names></string-name>, <string-name><surname>Himakunthala</surname> <given-names>V</given-names></string-name>, <string-name><surname>Ouyang</surname> <given-names>A</given-names></string-name>, <string-name><surname>He</surname> <given-names>R</given-names></string-name>, <string-name><surname>Mei</surname> <given-names>A</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Visual chain of thought: bridging logical gaps with multimodal infillings</article-title>. <comment>arXiv:2305.02317. 2024</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2305.02317</pub-id>.</mixed-citation></ref>
<ref id="ref-120"><label>[120]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Cai</surname> <given-names>K</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>X</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Lv</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>R</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>MM-CoT: a benchmark for probing visual chain-of-thought reasoning in multimodal models</article-title>. <comment>arXiv:2512.08228. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2512.08228</pub-id>.</mixed-citation></ref>
<ref id="ref-121"><label>[121]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhou</surname> <given-names>D</given-names></string-name>, <string-name><surname>Sch&#x00E4;rli</surname> <given-names>N</given-names></string-name>, <string-name><surname>Hou</surname> <given-names>L</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>J</given-names></string-name>, <string-name><surname>Scales</surname> <given-names>N</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Least-to-most prompting enables complex reasoning in large language models</article-title>. In: <conf-name>Proceedings of the Eleventh International Conference on Learning Representations; 2023 May 1&#x2013;5</conf-name>; <publisher-loc>Kigali, Rwanda</publisher-loc>.</mixed-citation></ref>
<ref id="ref-122"><label>[122]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Li</surname> <given-names>T</given-names></string-name>, <string-name><surname>Qin</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>L</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>C</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>UniGeo: unifying geometry logical reasoning via reformulating mathematical expression</article-title>. <comment>arXiv:2212.02746. 2022</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2212.02746</pub-id>.</mixed-citation></ref>
<ref id="ref-123"><label>[123]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Amini</surname> <given-names>A</given-names></string-name>, <string-name><surname>Gabriel</surname> <given-names>S</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>S</given-names></string-name>, <string-name><surname>Koncel-Kedziorski</surname> <given-names>R</given-names></string-name>, <string-name><surname>Choi</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Hajishirzi</surname> <given-names>H</given-names></string-name></person-group>. <article-title>MathQA: towards interpretable math word problem solving with operation-based formalisms</article-title>. In: <conf-name>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers); 2019 Jun 2&#x2013;7</conf-name>; <publisher-loc>Minneapolis, MN, USA</publisher-loc>. p. <fpage>2357</fpage>&#x2013;<lpage>67</lpage>.</mixed-citation></ref>
<ref id="ref-124"><label>[124]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>TE</given-names></string-name>, <string-name><surname>Lv</surname> <given-names>A</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Masked thought: simply masking partial reasoning steps can improve mathematical reasoning learning of language models</article-title>. In: <conf-name>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers); 2024 Aug 11&#x2013;16</conf-name>; <publisher-loc>Bangkok, Thailand</publisher-loc>. p. <fpage>5872</fpage>&#x2013;<lpage>900</lpage>.</mixed-citation></ref>
<ref id="ref-125"><label>[125]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Cobbe</surname> <given-names>K</given-names></string-name>, <string-name><surname>Kosaraju</surname> <given-names>V</given-names></string-name>, <string-name><surname>Bavarian</surname> <given-names>M</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>M</given-names></string-name>, <string-name><surname>Jun</surname> <given-names>H</given-names></string-name>, <string-name><surname>Kaiser</surname> <given-names>L</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Training verifiers to solve math word problems</article-title>. <comment>arXiv:2110.14168. 2021</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2110.14168</pub-id>.</mixed-citation></ref>
<ref id="ref-126"><label>[126]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>P</given-names></string-name>, <string-name><surname>Li</surname> <given-names>L</given-names></string-name>, <string-name><surname>Shao</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>R</given-names></string-name>, <string-name><surname>Dai</surname> <given-names>D</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Math-Shepherd: verify and reinforce LLMs step-by-step without human annotations</article-title>. In: <conf-name>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers); 2024 Aug 11&#x2013;16</conf-name>; <publisher-loc>Bangkok, Thailand</publisher-loc>. p. <fpage>9426</fpage>&#x2013;<lpage>39</lpage>.</mixed-citation></ref>
<ref id="ref-127"><label>[127]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Guo</surname> <given-names>D</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Song</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>P</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>Q</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning</article-title>. <source>Nature</source>. <year>2025</year>;<volume>645</volume>(<issue>8081</issue>):<fpage>633</fpage>&#x2013;<lpage>8</lpage>. doi:<pub-id pub-id-type="doi">10.1038/s41586-025-09422-z</pub-id>; <pub-id pub-id-type="pmid">40962978</pub-id></mixed-citation></ref>
<ref id="ref-128"><label>[128]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Azerbayev</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Schoelkopf</surname> <given-names>H</given-names></string-name>, <string-name><surname>Paster</surname> <given-names>K</given-names></string-name>, <string-name><surname>Santos</surname> <given-names>MD</given-names></string-name>, <string-name><surname>McAleer</surname> <given-names>SM</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>AQ</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Llemma: an open language model for mathematics</article-title>. In: <conf-name>Proceedings of the Twelfth International Conference on Learning Representations; 2024 May 7&#x2013;11</conf-name>; <publisher-loc>Vienna, Austria</publisher-loc>.</mixed-citation></ref>
<ref id="ref-129"><label>[129]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Mathew</surname> <given-names>M</given-names></string-name>, <string-name><surname>Karatzas</surname> <given-names>D</given-names></string-name>, <string-name><surname>Jawahar</surname> <given-names>CV</given-names></string-name></person-group>. <article-title>DocVQA: a dataset for VQA on document images</article-title>. In: <conf-name>Proceedings of the 2021 IEEE Winter Conference on Applications of Computer Vision (WACV); 2021 Jan 3&#x2013;8</conf-name>; <publisher-loc>Waikoloa, HI, USA</publisher-loc>. p. <fpage>2199</fpage>&#x2013;<lpage>208</lpage>. doi:<pub-id pub-id-type="doi">10.1109/wacv48630.2021.00225</pub-id>.</mixed-citation></ref>
<ref id="ref-130"><label>[130]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Masry</surname> <given-names>A</given-names></string-name>, <string-name><surname>Long</surname> <given-names>DX</given-names></string-name>, <string-name><surname>Tan</surname> <given-names>JQ</given-names></string-name>, <string-name><surname>Joty</surname> <given-names>S</given-names></string-name>, <string-name><surname>Hoque</surname> <given-names>E</given-names></string-name></person-group>. <article-title>ChartQA: a benchmark for question answering about charts with visual and logical reasoning</article-title>. In: <conf-name>Proceedings of the Findings of the Association for Computational Linguistics: ACL 2022; 2022 May 22&#x2013;27</conf-name>; <publisher-loc>Dublin, Ireland</publisher-loc>. p. <fpage>2263</fpage>&#x2013;<lpage>79</lpage>.</mixed-citation></ref>
<ref id="ref-131"><label>[131]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Fu</surname> <given-names>C</given-names></string-name>, <string-name><surname>Dai</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>L</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Video-MME: the first-ever comprehensive evaluation benchmark of multi-modal LLMs in video analysis</article-title>. In: <conf-name>Proceedings of the 2025 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2025 Jun 10&#x2013;17</conf-name>; <publisher-loc>Nashville, TN, USA</publisher-loc>. p. <fpage>24108</fpage>&#x2013;<lpage>18</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr52734.2025.02245</pub-id>.</mixed-citation></ref>
<ref id="ref-132"><label>[132]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>F</given-names></string-name>, <string-name><surname>Fu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>JY</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>MuirBench: a comprehensive benchmark for robust multi-image understanding</article-title>. In: <conf-name>Proceedings of the Thirteenth International Conference on Learning Representations; 2025 Apr 24&#x2013;28</conf-name>; <publisher-loc>Singapore, Singapore</publisher-loc>.</mixed-citation></ref>
<ref id="ref-133"><label>[133]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Gervais</surname> <given-names>P</given-names></string-name>, <string-name><surname>Fadeeva</surname> <given-names>A</given-names></string-name>, <string-name><surname>Maksai</surname> <given-names>A</given-names></string-name></person-group>. <article-title>MathWriting: a dataset for handwritten mathematical expression recognition</article-title>. In: <conf-name>Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2; 2025 Aug 3&#x2013;7</conf-name>; <publisher-loc>Toronto, ON, Canada</publisher-loc>. p. <fpage>5459</fpage>&#x2013;<lpage>69</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3711896.3737436</pub-id>.</mixed-citation></ref>
<ref id="ref-134"><label>[134]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Qi</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zeng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Bao</surname> <given-names>X</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>L</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>VCR-Bench: a comprehensive evaluation framework for video chain-of-thought reasoning</article-title>. <comment>arXiv:2504.07956. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2504.07956</pub-id>.</mixed-citation></ref>
<ref id="ref-135"><label>[135]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Brown</surname> <given-names>TB</given-names></string-name>, <string-name><surname>Mann</surname> <given-names>B</given-names></string-name>, <string-name><surname>Ryder</surname> <given-names>N</given-names></string-name>, <string-name><surname>Subbiah</surname> <given-names>M</given-names></string-name>, <string-name><surname>Kaplan</surname> <given-names>J</given-names></string-name>, <string-name><surname>Dhariwal</surname> <given-names>P</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Language models are few-shot learners</article-title>. In: <conf-name>Proceedings of the 34th International Conference on Neural Information Processing Systems NIPS&#x2019;20; 2020 Dec 6&#x2013;12</conf-name>; <publisher-loc>Virtual</publisher-loc>.</mixed-citation></ref>
<ref id="ref-136"><label>[136]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Huang</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>H</given-names></string-name></person-group>. <article-title>3D-R1: enhancing reasoning in 3D VLMs for unified scene understanding</article-title>. <comment>arXiv:2507.23478. 2025</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2507.23478</pub-id>.</mixed-citation></ref>
<ref id="ref-137"><label>[137]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Ji</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>WebSRC: a dataset for web-based structural reading comprehension</article-title>. In: <conf-name>Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing; 2021 Nov 7&#x2013;11; Online</conf-name>. p. <fpage>4073</fpage>&#x2013;<lpage>85</lpage>.</mixed-citation></ref>
<ref id="ref-138"><label>[138]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Jaegle</surname> <given-names>A</given-names></string-name>, <string-name><surname>Gimeno</surname> <given-names>F</given-names></string-name>, <string-name><surname>Brock</surname> <given-names>A</given-names></string-name>, <string-name><surname>Vinyals</surname> <given-names>O</given-names></string-name>, <string-name><surname>Zisserman</surname> <given-names>A</given-names></string-name>, <string-name><surname>Carreira</surname> <given-names>J</given-names></string-name></person-group>. <chapter-title>Perceiver: general perception with iterative attention</chapter-title>. In: <person-group person-group-type="editor"><string-name><surname>Meila</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>T</given-names></string-name></person-group>, editors. <source>Proceedings of the 38th International Conference on Machine Learning; 2021 Jul 18&#x2013;24</source>; <publisher-loc>Virtual</publisher-loc>. p. <fpage>4651</fpage>&#x2013;<lpage>64</lpage>.</mixed-citation></ref>
</ref-list>
</back></article>