<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1752580</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Technology and Code</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Interpretable multimodal reasoning for robo-advisory: the FinErva framework</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Chi</surname> <given-names>Jiarui</given-names></name>
<xref ref-type="aff" rid="aff1"/>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<uri xlink:href="https://loop.frontiersin.org/people/3223078"/>
</contrib>
</contrib-group>
<aff id="aff1"><institution>PBC School of Finance, Tsinghua University</institution>, <city>Beijing</city>, <country country="cn">China</country></aff>
<author-notes>
<corresp id="c001"><label>&#x0002A;</label>Correspondence: Jiarui Chi, <email xlink:href="mailto:jerrychi2004@gmail.com">jerrychi2004@gmail.com</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-01-21">
<day>21</day>
<month>01</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1752580</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>11</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>04</day>
<month>12</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>12</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2026 Chi.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>Chi</copyright-holder>
<license>
<ali:license_ref start_date="2026-01-21">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>The rapid development of robo-advisory and quantitative investment has been accompanied by persistent concerns about limited personalization and the opacity of black-box models operating on multimodal financial information. This paper addresses these issues from a decision-support perspective by constructing FinErva, a multimodal chain-of-thought dataset tailored to financial applications. FinErva comprises 7,544 manually verified question&#x02013;answer pairs, divided into two economically relevant tasks: contract and disclosure understanding (FinErva-Pact) and candlestick-chart-based technical analysis (FinErva-Price). Building on this dataset, the paper propose a two-stage training framework: Supervised-CoT Learning followed by Self-CoT Refinement, and apply it to eight vision&#x02013;language models, each with fewer than 0.8 billion parameters. Empirical results show that those lightweight models approach the performance of finance professionals and clearly outperform non-expert investors. Overall, the findings indicate that appropriately designed multimodal chain of thought supervision enables interpretable modeling of key research tasks such as contract review and chart interpretation under realistic computational and deployment constraints, providing new data and methodology for the development of personalized, explainable, and operationally feasible AI systems in investment advisory and risk management.</p></abstract>
<kwd-group>
<kwd>chain-of-thought</kwd>
<kwd>explainable artificial intelligence</kwd>
<kwd>investment decision support</kwd>
<kwd>lightweight and low cost</kwd>
<kwd>multimodal financial reasoning</kwd>
<kwd>robo-advisory</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declared that financial support was not received for this work and/or its publication.</funding-statement>
</funding-group>
<counts>
<fig-count count="3"/>
<table-count count="8"/>
<equation-count count="13"/>
<ref-count count="54"/>
<page-count count="0"/>
<word-count count="10218"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>AI in Finance</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>In recent years, robo-advisors have gradually emerged as a central form of retail wealth management (<xref ref-type="bibr" rid="B14">Eskandarany, 2024</xref>; <xref ref-type="bibr" rid="B44">Theodorakopoulos et al., 2025</xref>) and are widely regarded as a key technology for replacing human financial advisors (<xref ref-type="bibr" rid="B22">Jadhav and Mirza, 2025</xref>) with quantitative models, reducing service costs, and improving advisory efficiency (<xref ref-type="bibr" rid="B40">Sutiene et al., 2024</xref>). However, in practice there remains a substantial gap between robo-advisors and experienced human analysts in terms of personalization capabilities and decision quality in complex scenarios (<xref ref-type="bibr" rid="B22">Jadhav and Mirza, 2025</xref>). Existing studies and industry reports (<xref ref-type="bibr" rid="B46">Verma et al., 2025</xref>; <xref ref-type="bibr" rid="B23">Jung et al., 2018</xref>; <xref ref-type="bibr" rid="B19">Goswami et al., 2025</xref>) indicate that most current robo-advisors still rely on low-dimensional risk-preference questionnaires and pre-specified model portfolios (<xref ref-type="bibr" rid="B3">Boreiko and Massarotti, 2020</xref>), which are insufficient to capture investors&#x00027; heterogeneous preferences and behavioral characteristics. As a result, the recommended portfolios tend to exhibit a &#x0201C;one-size-fits-all&#x0201D; pattern, and their decision performance often falls short of that of seasoned professional analysts (<xref ref-type="bibr" rid="B11">D&#x00027;Acunto et al., 2019</xref>). This perception of &#x0201C;insufficient personalization and questionable decision quality&#x0201D; is one of the fundamental reasons why investors remain cautious about adopting robo-advisory services (<xref ref-type="bibr" rid="B46">Verma et al., 2025</xref>).</p>
<p>From an institutional perspective, achieving true personalization and specialization in stock investment decision-making and trading strategy research often requires the construction of proprietary models tailored to specific markets, asset classes, or even investment styles. These proprietary models not only demand substantial feature engineering and parameter tuning during the research and development phase, but also face a series of cost constraints (<xref ref-type="bibr" rid="B10">Cottier et al., 2025</xref>; <xref ref-type="bibr" rid="B31">Maple et al., 2024</xref>; <xref ref-type="bibr" rid="B21">Interpress, 2024</xref>), such as computational power, storage, latency control, and regulatory scrutiny, during the deployment phase (<xref ref-type="bibr" rid="B32">Paleyes et al., 2022</xref>; <xref ref-type="bibr" rid="B37">Sen et al., 2021</xref>). As the scale of models rapidly expands, the parameter size of large-scale pre-trained models has progressed from tens of billions to hundreds of billions, leading to a sharp increase in the computational power and data required for training and fine-tuning. This has made it increasingly difficult in practice to customize large models for a single institution or a single strategy. Even within relatively traditional machine learning frameworks, the introduction of methods such as ensemble learning and deep networks significantly raises the costs of model development and maintenance, not to mention the added complexity of incorporating language models and multimodal models on top of these approaches.</p>
<p>On the quantitative modeling side, most decision and regression frameworks used in investment management are still grounded in linear or generalized linear models (<xref ref-type="bibr" rid="B24">Kwon, 2025</xref>; <xref ref-type="bibr" rid="B27">Liu and Song, 2025</xref>; <xref ref-type="bibr" rid="B15">Feng et al., 2025</xref>; <xref ref-type="bibr" rid="B41">Tan et al., 2025</xref>) from classical factor models to regularized regressions and generalized linear risk models. While these approaches have delivered tractable estimation procedures, they are structurally limited in capturing the complex, nonlinear interactions and regime-dependent patterns that characterize modern financial markets. Empirical asset-pricing research (<xref ref-type="bibr" rid="B1">Bagnara, 2024</xref>; <xref ref-type="bibr" rid="B6">Chen et al., 2024</xref>) demonstrate that nonlinear machine learning models, such as tree ensembles and deep neural networks, are able to extract economically meaningful signals from high-dimensional characteristics and often generate substantial improvements in out-of-sample Sharpe ratios compared to leading linear benchmarks. Building upon this, Large Language Models (LLMs) take it a step further by extending the &#x0201C;high-dimensional pattern recognition&#x0201D; ability to unstructured text and even multimodal (text modal and vision modal) data. Through pre-training and transfer learning on large-scale corpora within the financial context, LLMs achieve effective optimization on highly non-convex objective functions. Even when faced with extremely high-dimensional decision spaces, they exhibit superior expressiveness and robustness compared to traditional linear frameworks.</p>
<p>However, simply relying on stronger pattern recognition abilities is not enough to drive the large-scale deployment of robo-advisors in real-world financial scenarios. For high-risk, heavily regulated financial businesses, the model&#x00027;s interpretability and auditability are just as important as predictive accuracy (<xref ref-type="bibr" rid="B30">Maier et al., 2022</xref>; <xref ref-type="bibr" rid="B4">Bussmann et al., 2020</xref>; <xref ref-type="bibr" rid="B17">Fritz-Morgenthal et al., 2022</xref>). On one hand, regulatory bodies and compliance departments need to trace and hold the model&#x00027;s decision-making logic accountable. On the other hand, end investors are more likely to entrust real wealth to automated systems if they understand how the model arrived at this asset allocation or trading recommendation. In the frontier of large model research (firstly introduced by <xref ref-type="bibr" rid="B49">Wei et al., 2023</xref>), the Chain-of-Thought (CoT) prompt has been proposed as a paradigm that allows the model to explicitly display its intermediate reasoning process: by guiding the model to generate step-by-step reasoning before providing the final conclusion, CoT not only significantly improves accuracy on complex reasoning tasks but also provides a direct entry point for human review and intervention in model decisions. Subsequent research (<xref ref-type="bibr" rid="B53">Zhang et al., 2024</xref>) has extended CoT to multimodal scenarios, showing that when jointly processing text and images, having the model output a structured reasoning chain can effectively align visual evidence with linguistic reasoning, thus enhancing performance and interpretability in multimodal question-answering and decision-making tasks. For robo-advisors, an interpretable Chain-of-Thought means that the model not only provides a &#x0201C;buy/sell/hold&#x0201D; conclusion but also clearly points to the underlying price trends, technical patterns, financial indicators, and textual information, thereby enhancing both decision accuracy and investor trust.</p>
<p>At the same time, real-world investment decisions are inherently multimodal. A human analyst typically integrates structured market and fundamental data (time series, ratios, and factor exposures), unstructured textual information (earnings calls, news, analyst reports), and visual signals (candlestick charts, technical indicators, and even screenshots of trading interfaces). However, to the best of our knowledge, research on &#x0201C;multimodal &#x0002B; Chain-of-Thought &#x0002B; large model&#x0201D; systems specifically tailored for investment decision-making scenarios remains scarce. Existing work (shown in <xref ref-type="table" rid="T1">Table 1</xref>) either lacks clear intermediate reasoning annotations or focuses solely on a single modality or task, making it difficult to comprehensively support the multimodal decision-making process modeling required for intelligent robo-advisors.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Comparison of existing finance datasets: &#x02717;* means text-extracted-format data, &#x02713;* means partial correct.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Domain</bold></th>
<th valign="top" align="left"><bold>Sample in Fin</bold></th>
<th valign="top" align="center"><bold>MultiModal</bold></th>
<th valign="top" align="left"><bold>Data description</bold></th>
<th valign="top" align="center"><bold>CoT</bold></th>
<th valign="top" align="center"><bold>GT</bold></th>
<th valign="top" align="left"><bold>Language</bold></th>
<th valign="top" align="left"><bold>Reasoning only from image</bold></th>
<th valign="top" align="left"><bold>Image complexity</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Ant_Finance (Team A., <xref ref-type="bibr" rid="B42">2023</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">13K</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="left">Understanding and reasoning</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02713;*</td>
<td valign="top" align="left">ZH</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">0</td>
</tr>
<tr>
<td valign="top" align="left">FinanceIQ (Team D. D., <xref ref-type="bibr" rid="B43">2023</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">Large enough</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="left">Open-domain question</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="left">ZH</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">0</td>
</tr>
<tr>
<td valign="top" align="left">FinQA (<xref ref-type="bibr" rid="B7">Chen et al., 2022a</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">8.3K</td>
<td valign="top" align="center">&#x02717;*</td>
<td valign="top" align="left">Understanding and reasoning</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">EN</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">0</td>
</tr>
<tr>
<td valign="top" align="left">Finance-Instruct (<xref ref-type="bibr" rid="B16">Flowers, 2025</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">500K</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="left">Reasoning, sentiment analysis</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02713;*</td>
<td valign="top" align="left">Mul-lang</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">0</td>
</tr>
<tr>
<td valign="top" align="left">BBF-Fin (<xref ref-type="bibr" rid="B29">Lu et al., 2023</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">Large enough</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="left">Understanding and generation</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="left">ZH</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">0</td>
</tr>
<tr>
<td valign="top" align="left">MME-Finance (<xref ref-type="bibr" rid="B18">Gan et al., 2024</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">1.2K</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">Open-ended question</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="left">ZH,EN</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">3</td>
</tr>
<tr>
<td valign="top" align="left">ConvFinQA (<xref ref-type="bibr" rid="B8">Chen et al., 2022b</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">3.9K</td>
<td valign="top" align="center">&#x02717;*</td>
<td valign="top" align="left">Table understanding</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">EN</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">0</td>
</tr>
<tr>
<td valign="top" align="left">TAT-QA (<xref ref-type="bibr" rid="B54">Zhu et al., 2021</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">16.5K</td>
<td valign="top" align="center">&#x02717;*</td>
<td valign="top" align="left">Numerical reasoning</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02713;*</td>
<td valign="top" align="left">EN</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">0</td>
</tr>
<tr>
<td valign="top" align="left">FAMMA (<xref ref-type="bibr" rid="B52">Xue et al., 2025</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">1.9K</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">Understanding and reasoning</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">EN,FR</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">2</td>
</tr>
<tr>
<td valign="top" align="left">Fin-Fact (<xref ref-type="bibr" rid="B35">Rangapur et al., 2024</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">3.6K</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">Fact judgment</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">EN</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">1</td>
</tr>
<tr>
<td valign="top" align="left">PDF-VQA (<xref ref-type="bibr" rid="B12">Ding et al., 2023</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">140K</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">PDF understanding</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02713;*</td>
<td valign="top" align="left">EN</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">3</td>
</tr>
<tr>
<td valign="top" align="left">Sujet-finance-QA (<xref ref-type="bibr" rid="B39">Sujet and Allaa Boutaleb, 2024</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">100K</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">Understanding and generation</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">EN</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">4,5</td>
</tr>
<tr>
<td valign="top" align="left">FinVis-GPT (<xref ref-type="bibr" rid="B47">Wang et al., 2023</xref>)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">1M</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">Open-ended question</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="center">&#x02717;</td>
<td valign="top" align="left">ZH,EN</td>
<td valign="top" align="left">&#x02717;</td>
<td valign="top" align="left">3</td>
</tr>
<tr>
<td valign="top" align="left">FinErva (this work)</td>
<td valign="top" align="center">Fin</td>
<td valign="top" align="left">7.54K</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">Understanding and reasoning</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="left">EN</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">3,4,5</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Image complexity from zero to five respectively indicates: no image, simple caption, structured tables, articles, or candlestick charts, images with noise such as handwriting and scan artifacts, highly noisy and semantically complex images that are difficult for humans to interpret. A red cross indicates that the criteria are not met, while a green checkmark indicates that the criteria are met.</p>
</table-wrap-foot>
</table-wrap>
<p>Addressing the aforementioned research gap, this paper proposes and develops the FinErva (FINancial-llm-with-minERVA<xref ref-type="fn" rid="fn0003"><sup>1</sup></xref>-wisdom) framework, which aims to provide a systematic data foundation and a lightweight model solution for multimodal Chain-of-Thought research in the field of intelligent robo-advisors. Specifically, FinErva integrates three representative financial scenarios&#x02014;financial contract and document understanding, real-world financial image interpretation, and technical analysis based on candlestick charts&#x02014;to create the first multimodal Chain-of-Thought question-answer dataset for the financial domain. Each sample contains real financial images, carefully designed question-answer pairs, multiple-choice options with distractors, and manually verified step-by-step reasoning chains. Based on this dataset, this paper further proposes a lightweight fine-tuning pipeline: by incorporating vision feature extraction modules at the visual encoding layer to address the visual encoding issues of lightweight models, and performing two-stage CoT fine-tuning on a series of open-source vision-language models with parameter sizes under 0.8B. The first stage involves supervised Chain-of-Thought learning (Supervised-CoT-Learning), while the second stage focuses on model self-refinement (Self-CoT-Learning), thus enabling multimodal reasoning and interpretability for financial tasks while keeping deployment costs under control. The key contributions of this work are as follows:</p>
<list list-type="bullet">
<list-item><p>The first multimodal Chain-of-Thought dataset and task setting specifically designed for the financial domain: The FinErva system systematically covers key scenarios such as financial contract understanding, complex financial scene image analysis, and technical analysis of candlestick chart patterns. By characterizing multimodal question-answering and reasoning requirements within a unified framework, FinErva provides a high-quality data foundation for subsequent research on intelligent robo-advisors and large financial models.</p></list-item>
<list-item><p>A reproducible and scalable low-cost, lightweight financial multimodal CoT fine-tuning pipeline: High-performance financial reasoning capabilities are achieved on vision-language models with parameter sizes under 0.8B. Experimental results show that the model fine-tuned on FinErva significantly outperforms both zero-shot in terms of accuracy. Additionally, the performance metrics approach or even exceed those of human experts with professional backgrounds (<xref ref-type="table" rid="T2">Table 2</xref>).</p></list-item>
<list-item><p>From the perspective of &#x0201C;interpretable intelligent robo-advisors,&#x0201D; this paper organically integrates CoT, LLMs, and multimodal financial scenarios: Through explicit Chain-of-Thought outputs and a low-cost deployment solution, FinErva provides a feasible pathway for constructing the next generation of robo-advisory systems that are both interpretable and capable of multimodal perception. It also lays the methodological foundation for auditable AI in application scenarios such as financial regulation, compliance review, and risk management.</p></list-item>
</list>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Human evaluation accuracy.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Human evaluation</bold></th>
<th valign="top" align="center"><bold>Acc-pact</bold></th>
<th valign="top" align="center"><bold>Acc-price</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Finance expert 1</td>
<td valign="top" align="center">72%</td>
<td valign="top" align="center">88%</td>
</tr>
<tr>
<td valign="top" align="left">Finance expert 2</td>
<td valign="top" align="center">64%</td>
<td valign="top" align="center">82%</td>
</tr>
<tr>
<td valign="top" align="left">Random participant 1</td>
<td valign="top" align="center">20%</td>
<td valign="top" align="center">56%</td>
</tr>
<tr>
<td valign="top" align="left">Random participant 2</td>
<td valign="top" align="center">32%</td>
<td valign="top" align="center">62%</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Fine-tuned model (ours)</bold></td>
<td valign="top" align="center"><bold>68.29%</bold></td>
<td valign="top" align="center"><bold>86.03%</bold></td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<p>Large language models and generative AI have recently emerged as flexible building blocks for financial analytics, including portfolio optimization, risk management, algorithmic trading, robo-advisory, and ESG analytics. Studies consistently report that deep learning and LLM-based systems can improve predictive performance or reduce information-processing costs, but also that they introduce new challenges in terms of explainability, fairness, operational risk, and infrastructure investment. This section primarily discusses the current state of research in the various fields relevant to this study, and Sections 2.4, 2.5 respectively outline the technical value and financial significance.</p>
<sec>
<label>2.1</label>
<title>Financial QA and document/table understanding</title>
<p>Studies focuse on financial QA and document or table understanding, where models are trained to answer questions about financial reports, prospectuses, or numerical tables. FinQA (<xref ref-type="bibr" rid="B7">Chen et al., 2022a</xref>) and ConvFinQA (<xref ref-type="bibr" rid="B8">Chen et al., 2022b</xref>) center on numerical reasoning over financial tables and textual contexts, simulating analyst-style questions based on financial documents. TAT-QA (<xref ref-type="bibr" rid="B54">Zhu et al., 2021</xref>) extends this paradigm to more complex table&#x02013;text interactions, where answers require multi-step aggregation and cross-referencing between text and semi-structured tables. FinanceIQ (Team D. D., <xref ref-type="bibr" rid="B43">2023</xref>) and Finance-Instruct (<xref ref-type="bibr" rid="B16">Flowers, 2025</xref>) provide large-scale instruction-style QA corpora for financial knowledge and task-oriented dialogue, while BBF-Fin (<xref ref-type="bibr" rid="B29">Lu et al., 2023</xref>) targets Chinese-language financial understanding and generation. Collectively, these benchmarks have enabled substantial progress in text-based and table-based financial reasoning. On the benchmarking side, FinBen (<xref ref-type="bibr" rid="B51">Xie et al., 2024</xref>) is proposed as a holistic financial benchmark for LLMs, covering a wide range of tasks including factual knowledge, numerical reasoning, and document understanding in finance. Complementary work such as Fino1 (<xref ref-type="bibr" rid="B33">Qian et al., 2025</xref>) and Fin-R1 (<xref ref-type="bibr" rid="B28">Liu et al., 2025</xref>) explores how reasoning-enhanced LLMs and reinforcement-learning-based alignment can improve financial question answering and reasoning quality on text-only financial tasks.</p>
<p>However, most of these studies share three structural limitations. First, they operate primarily on structured or semi-structured inputs (tables and machine-readable text) and thus under-represent realistic financial artifacts such as scanned contracts, handwritten annotations, or chart screenshots. Second, they partially provide final answers (Ground Truth), but not explicit, human-authored reasoning trajectories that could be used to train or evaluate chain-of-thought explanations. Third, the target tasks are usually framed as isolated QA problems rather than as components of a broader, multimodal investment decision process.</p>
</sec>
<sec>
<label>2.2</label>
<title>Multimodal financial datasets and vision&#x02013;language benchmarks</title>
<p>To bridge the gap between textual financial QA and the rich visual environment of practical investing, several multimodal or vision&#x02013;language datasets have been proposed. MME-Finance (<xref ref-type="bibr" rid="B18">Gan et al., 2024</xref>) extends this landscape by introducing a relatively small bilingual multimodal dataset with open-ended questions over financial images, while FAMMA (<xref ref-type="bibr" rid="B52">Xue et al., 2025</xref>) focuses on multilingual multimodal QA, including French and English text over financial documents. Fin-Fact (<xref ref-type="bibr" rid="B35">Rangapur et al., 2024</xref>) contributes a multimodal fact-checking benchmark that combines textual claims with evidence from financial images, and PDF-VQA (<xref ref-type="bibr" rid="B12">Ding et al., 2023</xref>) targets visual question answering over noisy, real-world PDF documents. Sujet-Finance-QA-Vision-100k (<xref ref-type="bibr" rid="B39">Sujet and Allaa Boutaleb, 2024</xref>) scales document VQA to 100k financial samples, and FinVis-GPT (<xref ref-type="bibr" rid="B47">Wang et al., 2023</xref>) introduces a multimodal LLM specifically for financial chart analysis.</p>
<p>These studies demonstrate that multimodal financial reasoning is both technically feasible and practically valuable. However, when examined from the perspective of interpretable financial reasoning, existing resources remain clearly insufficient. First, most datasets cover only a single category of visual objects (e.g., charts or PDFs), making it difficult to jointly model heterogeneous information sources such as contracts, real-world financial scenes, and candlestick charts within a unified framework. Second, they generally lack explicit human-annotated chain of thought reasoning, which prevents direct training and evaluation of CoT models that align textual reasoning with visual evidence. Third, existing task formulations rarely target the full investment decision making process&#x02014;such as deriving step-by-step investment conclusions from contract terms or market scenes&#x02014;ultimately leading to actionable trading or allocation recommendations. Consequently, these datasets can support only isolated capabilities of multimodal robo-advisors but fall short of enabling an end-to-end, interpretable multimodal decision-making system.</p>
</sec>
<sec>
<label>2.3</label>
<title>Chain-of-thought and multimodal reasoning</title>
<p>In the broader AI literature, chain-of-thought prompting has emerged as a simple yet powerful technique for improving the reasoning capabilities of LLMs. <xref ref-type="bibr" rid="B49">Wei et al. (2023)</xref> show that providing a small number of demonstrations with explicit intermediate reasoning steps can dramatically enhance performance on arithmetic, commonsense, and symbolic reasoning benchmarks. Subsequently, a growing body of research (<xref ref-type="bibr" rid="B5">Chellappa et al., 2024</xref>; <xref ref-type="bibr" rid="B38">Shao et al., 2024</xref>; <xref ref-type="bibr" rid="B20">Hegde et al., 2025</xref>; <xref ref-type="bibr" rid="B25">Leong et al., 2024</xref>) has incorporated CoT reasoning into multimodal settings.</p>
<p>However, existing multimodal CoT datasets are domain-general and do not capture the specific semantics and constraints of financial decision-making. There is, to the best of our knowledge, no publicly available dataset that combines real-world financial images (contracts, market scenes, and candlestick charts) with high-quality, human-verified chain of thought annotations specifically tailored to investment and advisory tasks.</p>
</sec>
<sec>
<label>2.4</label>
<title>Identified research gaps and FinErva</title>
<p>Synthesizing the above literature, this study contributes to both the AI and finance domains. First, compared with existing text-only financial QA benchmarks and multimodal datasets, FinErva is the first dataset that simultaneously covers financial contract understanding, real-world financial scene interpretation, and candlestick-based technical analysis, while providing multimodal chain of thought annotations within a unified framework. Second, in contrast to general purpose multimodal CoT benchmarks, FinErva&#x00027;s tasks and annotations are specifically designed for financial decision scenarios, embedding concepts, such as order types, corporate actions, and chart patterns&#x02014;directly into the reasoning chains.</p>
<p>Third, building on this dataset, this paper proposes a lightweight fine-tuning pipeline for vision&#x02013;language models with fewer than 0.8 billion parameters, showing that such compact models can achieve expert-level performance on financial multimodal reasoning tasks when trained with appropriate CoT supervision. This direction aligns closely with the development of scalable, interpretable, and financial AI tools, and it directly addresses the practical constraints, such as computational cost, latency, and governance, that large-scale black-box models face when being deployed in production-level robo-advisory systems.</p>
</sec>
<sec>
<label>2.5</label>
<title>Financial significance and relevance</title>
<p>First, this paper contributes to the field of intelligent financial advisory not only by enhancing interpretability but also by improving the understanding of contracts/disclosures and technical analysis through candlestick charts, thereby optimizing the quality of investment advice and investor outcomes. Currently, a significant challenge faced by intelligent financial advisory systems is how to extract meaningful signals from vast amounts of financial data while providing personalized and compliant recommendations. By incorporating contract and disclosure understanding, this study helps investors identify potential legal risks, such as mis-selling or suitability violations, which is crucial for reducing losses due to information asymmetry or misleading information. Specifically, accurate comprehension of contractual terms and disclosure contents ensures that investment advice aligns with suitability standards, preventing legal disputes and safeguarding investor rights due to improper recommendations. Additionally, technical analysis based on K-line charts provides timely market trend signals, which are vital for risk management and portfolio construction, assisting investors in developing rational trading strategies and reducing emotion-driven investment decisions. The improvements in these two tasks fundamentally enhance the quality of investment advice from intelligent financial advisors, making it not only responsive to market demand but also protective of investors&#x00027; long-term interests.</p>
<p>More generally, the multimodal chain of thought framework proposed in this study is closely related to financial theory, particularly regulatory requirements such as fiduciary duty, suitability standards, and disclosure obligations in financial services. During the development of intelligent financial advisory systems, regulatory bodies require financial service providers to adhere to fiduciary responsibilities and suitability standards, meaning personalized investment advice must be based on clients&#x00027; risk tolerance, investment goals, and financial status. This standard mandates that intelligent advisory systems explicitly explain their decision-making process when formulating investment advice. The framework presented in this paper enhances the interpretability of intelligent advisory systems, allowing each piece of investment advice to be traced back to specific contract terms, market data, and technical signals, thus helping financial institutions comply with regulatory requirements and ensure investor rights are protected. On this basis, the research also offers new perspectives for financial regulation. By providing auditable reasoning chains, regulatory bodies can more transparently assess and supervise the decision-making processes of intelligent advisors, ensuring compliance with industry standards and supporting the development of future regulatory frameworks in the financial sector.</p>
<p>Also, this study not only provides contributions to the innovation of financial technologies in terms of data and methodologies but also defines the primary target users of the framework and dataset: intelligent financial advisory developers, academic researchers, and regulatory bodies. For intelligent financial advisory developers, the multimodal chain of thought framework and dataset constructed in this study provide new tools and methodologies, enabling them to deliver more personalized and compliant investment advice in complex financial data environments. For academic researchers, this study offers new data sources and research platforms for further optimization of intelligent advisory systems and financial decision-making research. Specifically, in areas such as risk management in intelligent advisory systems, investor behavior analysis, and compliance review, the application of the FinErva dataset will foster the interdisciplinary integration of finance and artificial intelligence, driving deeper academic exploration. For regulatory bodies, as financial technology evolves rapidly, ensuring that these technologies comply with regulatory requirements and protect investor interests has become an important issue. The framework and dataset provided by this study will assist regulatory bodies in the supervision and assessment of intelligent advisory systems, particularly in ensuring transparency, interpretability, and compliance of investment recommendations, thereby providing a theoretical foundation and practical tools for sustainable development and financial technology regulation in the financial sector.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>The FinErva dataset</title>
<sec>
<label>3.1</label>
<title>Overview</title>
<p><bold>FinErva</bold> is a multimodal financial question answering dataset designed to facilitate research on chain of thought reasoning based on both visual and textual modalities. It comprises <bold>7.54K multimodal samples</bold>, each data accompanied by: a real-world financial image (e.g., financial contracts, financial statements, candlestick charts, etc.); one correct answer and two distractors crafted to mislead large language models; a detailed caption describing the image content; and a chain-of-thought rationale for solving the corresponding question.</p>
<p><xref ref-type="table" rid="T3">Table 3</xref> demonstrates the split statistics of FinErva. FinErva consists of 7,544 samples divided into two dimensions: FinErva-Pact and FinErva-Price, each with corresponding training, validation, and test splits. FinErva-Pact contains 5,488 samples and is composed of real financial contract QA pairs, focusing on understanding and computation based on textual content within financial documents. FinErva-Price includes 2,056 samples, consisting of real financial candlestick chart QA pairs, targeting detailed interpretation and computation involving complex graphical patterns and quantitative reasoning over complex visual patterns, such as Stock market price analysis.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>FinErva statistics.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="left"><bold>Split</bold></th>
<th valign="top" align="center"><bold>Count</bold></th>
<th valign="top" align="left"><bold>Notes</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">FinErva-Pact</td>
<td valign="top" align="left">Train</td>
<td valign="top" align="center">3,841</td>
<td valign="top" align="left">Financial contract</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Test</td>
<td valign="top" align="center">823</td>
<td/>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Val</td>
<td valign="top" align="center">824</td>
<td/>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Total</td>
<td valign="top" align="center">5,488</td>
<td/>
</tr>
<tr>
<td valign="top" align="left">FinErva-Price</td>
<td valign="top" align="left">Train</td>
<td valign="top" align="center">1,440</td>
<td valign="top" align="left">Financial chart</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Test</td>
<td valign="top" align="center">308</td>
<td/>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Val</td>
<td valign="top" align="center">308</td>
<td/>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Total</td>
<td valign="top" align="center">2,056</td>
<td/>
</tr>
<tr>
<td valign="top" align="left" colspan="2"><bold>Overall total</bold></td>
<td valign="top" align="center"><bold>7,544</bold></td>
<td/>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<label>3.2</label>
<title>Annotation procedure</title>
<p>The annotation process is carefully designed by the author and executed by a professional annotation team. A standard annotation workflow is adopted, consisting of model-assisted pre-annotation followed by human verification. The author confirm that all AI-assisted tools used in this stage are fully legitimate and fully compliant with academic ethical standards.</p>
<sec>
<label>3.2.1</label>
<title>API annotation</title>
<p>First, we use OpenAI&#x00027;s API to generate two distractor choices, a query, and a chain-of-thought solution for each sample. We employ prompt templates tailored for <monospace>ChatGPT-o4-mini-high</monospace>, a model particularly strong in visual&#x02013;textual reasoning. The complete prompt templates are provided in Appendix B. To facilitate consistent task execution by large language models, the prompts explicitly instruct the model to always assign the correct answer to option A.</p>
<p>Subsequently, during post-processing, the answer options are randomly shuffled to ensure that the correct answers are evenly distributed among the three choices. Importantly, each sample is subsequently reviewed and verified through careful human annotation stage to ensure correctness.</p></sec>
<sec>
<label>3.2.2</label>
<title>Human annotation</title>
<p>Human annotation constitutes the core of our experiments and has been meticulously designed. Each annotator holds a Master&#x00027;s degree or higher and possesses a proficient level of English (annotators for whom English is not a native language have passed the university English proficiency test in their respective regions). They also have at least three years of professional knowledge in finance, including an understanding of financial contracts and candlestick charts, to ensure that every annotator has the necessary expertise. Each instance in the dataset was independently judged by two human annotators, so that every data point underwent two separate rounds of evaluation. Annotators were fairly paid by 28USD per 200 samples, and their participation was voluntary and conducted under responsible data using guidelines.</p>
<p>The specific guidelines are as follows: Annotators evaluate each data point across four dimensions, from top to bottom, as shown in the <xref ref-type="table" rid="T4">Table 4</xref>, Problem not valid, No valid option exists, Provided wrong answer, and Reasoning integrity. If an annotator believes that none of the first three dimensions (Problem not valid, No valid option exists, Provided wrong answer) contain errors, meaning all fields in the data do not show obvious mistakes, the annotator will score the reasoning chain (the evaluation of the Reasoning integrity dimension). The score ranges from 5 to 1, representing a spectrum from perfect (5 points) to unacceptable (1 point). If the score given by the annotator is less than 3, it is considered that the evaluation of the fourth dimension (Reasoning integrity) fails, meaning the reasoning chain, although free from obvious errors, does not fully simulate human thought processes. Each data is independently annotated by two annotators, and the data is considered annotated only when both annotators assign a score greater than 3 to the reasoning chain.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Human-checked error dimension.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Error dimension</bold></th>
<th valign="top" align="center"><bold>Proportion</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Problem not valid</td>
<td valign="top" align="center">43.72%</td>
</tr>
<tr>
<td valign="top" align="left">No valid option exists</td>
<td valign="top" align="center">12.28%</td>
</tr>
<tr>
<td valign="top" align="left">Provided wrong answer</td>
<td valign="top" align="center">18.89%</td>
</tr>
<tr>
<td valign="top" align="left">Reasoning integrity</td>
<td valign="top" align="center">9.65%</td>
</tr>
<tr>
<td valign="top" align="left">No obvious errors</td>
<td valign="top" align="center">15.46%</td>
</tr></tbody>
</table>
</table-wrap>
<p>Specifically, when annotators identify a clear error in the data, they will mark the erroneous dimension and provide a reasonable correction, including a correct question, answer, distractions and reasoning. This data will then be returned to the unannotated data queue to be reviewed by another annotator. If the corrected data is deemed no error, it will be sent to a third annotator for verification, and the data will be considered finalized only when at least two annotators approve it.</p>
<p>In practice, after correction by one annotator, the data should not contain obvious errors. A very small number of data may be contentious and will be discussed by the entire annotation team. Another common situation of inconsistency arises when one annotator does not find an obvious error and assigns a reasoning chain score, while another annotator believes there is an error, directly corrects it, and returns it to the unannotated queue. In such cases, the data must be approved by two additional annotators before it is considered finalized. While the process may seem complex, it is easy to implement in practice, as it only requires identifying data points approved by a single annotator and having them annotated by two others. It is important to note that the process of scoring the reasoning chain is highly subjective and cannot be constrained by visualizable rules. Therefore, annotation is considered complete when both annotators agree that the reasoning chain is complete.</p>
<p>Overall, this annotation guideline is entirely human-driven and is strictly enforced.</p></sec>
<sec>
<label>3.2.3</label>
<title>Data quality assessment</title>
<p>To verify the quality of the constructed dataset, multiple volunteers are recruited to participate in a human evaluation study. The volunteers are divided into two groups: finance majors and random participants, all of whom are graduate students. Samples are randomly drawn from the test set in Price and Pact, presented to the volunteers for manual answering. The results are summarized in <xref ref-type="table" rid="T2">Table 2</xref>. As shown in the table, the fine-tuned model significantly outperforms random participants and achieves comparable accuracy to that of finance professionals. A comprehensive quantitative evaluation of model accuracy is provided in Section 5. Although the results are subject to potential variance due to the small-sample randomness, they overall demonstrate both the domain expertise embodied in our dataset and the effectiveness of the fine-tuned model. One image is equipped with multiple questions ranging from different aspect. The training data and test data are disjointly split to ensure that no image or question in training and test set both, which prevents data leakage and enables a more reliable evaluation of model performance.</p>
<p>For better interpretability, we also visualize several representative examples from the dataset in <xref ref-type="fig" rid="F1">Figure 1</xref>, which illustrate the typical difficulty level and reasoning characteristics of our data samples.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>FinErva data. <bold>(a)</bold> A sample in FinErva-pact. <bold>(b)</bold> A sample in FinErva-price.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1752580-g0001.tif">
<alt-text content-type="machine-generated">An image with two labeled sections showing examples related to financial analysis. The first section (a) titled &#x0201C;A Sample in FinErva-Pact&#x0201D; presents a question about reducing a quarter's budget to minimize risks and optimize cash flow, with the chosen answer being option B. The explanation highlights Q3's high distribution at 33 percent as a reason for choosing it. The second section (b) titled &#x0201C;A Sample in FinErva-Price&#x0201D; includes a question on candlestick charts, with answer B involving higher lows and moderate volume expansion as reliable indicators. A chart displays price trends over time.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec>
<label>3.3</label>
<title>Data analysis</title>
<sec>
<label>3.3.1</label>
<title>Dataset source</title>
<p>FinErva comes from publicly available financial disclosure documents (such as financial reports, company announcements, stock market analysis reports, etc.), financial data service platforms, and historical K-line data from the financial markets (<xref ref-type="bibr" rid="B39">Sujet and Allaa Boutaleb, 2024</xref>; <xref ref-type="bibr" rid="B47">Wang et al., 2023</xref>)(random selected). All data is publicly sourced and does not involve any confidential or restricted data sources. Specifically, the dataset includes data from different markets, industries, and asset classes, ensuring the dataset&#x00027;s breadth and representativeness.</p></sec>
<sec>
<label>3.3.2</label>
<title>Data structure</title>
<p><xref ref-type="fig" rid="F1">Figure 1</xref> demonstrates one sample in FinErva-Pact and one sample in FinErva-Price. All questions in the dataset are formatted as multiple-choice, each with a single correct answer. Every question is deliberately constructed to require complete reliance on the visual input, ensuring that the model truly utilizes multimodal information. Additionally, each question is accompanied by a manually verified chain of thought rationale to guide model learning. The questions span various forms, including text understanding within images, numerical reasoning based on visual content, and analysis of real-world financial images. The overall goal is to equip models with strong visual reasoning capabilities in the financial domain. A complete example of the dataset is provided in the Appendix A. At the same time, we ensure the legality of all the data and guarantee that no sensitive or confidential information is involved and also ask readers to ensure under legal restrictions of their respective regions when using this dataset.</p></sec>
<sec>
<label>3.3.3</label>
<title>Data distribution</title>
<p>FinErva spans a comprehensive and diverse set of question types with varying levels of complexity, reflecting a typical retail investment advisory scenario. The Pact subset consists of simple question-answering tasks, whereas the Price subset focuses on complex understanding-reasoning tasks. <xref ref-type="table" rid="T3">Table 3</xref> shows detail size of FinErva. <xref ref-type="table" rid="T5">Tables 5</xref>, <xref ref-type="table" rid="T6">6</xref> report the distribution of question types across the two subsets, and we intentionally balance the number of questions in each category, ensuring that the capabilities learned by the model are comprehensive rather than biased toward any single modality or task. Training, test, validation sets are keeping the distribution.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Distribution of question types in the pact subset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Question type</bold></th>
<th valign="top" align="center"><bold>Proportion (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Numerical reasoning</td>
<td valign="top" align="center">31.82</td>
</tr>
<tr>
<td valign="top" align="left">Textual comprehension</td>
<td valign="top" align="center">37.15</td>
</tr>
<tr>
<td valign="top" align="left">Information retrieval</td>
<td valign="top" align="center">31.03</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Distribution of question types in the price subset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Question type</bold></th>
<th valign="top" align="center"><bold>Proportion (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Basic information comprehension</td>
<td valign="top" align="center">23.01</td>
</tr>
<tr>
<td valign="top" align="left">Basic numerical reasoning</td>
<td valign="top" align="center">21.98</td>
</tr>
<tr>
<td valign="top" align="left">Advanced computations</td>
<td valign="top" align="center">19.99</td>
</tr>
<tr>
<td valign="top" align="left">Technical-analysis tactics</td>
<td valign="top" align="center">19.02</td>
</tr>
<tr>
<td valign="top" align="left">Chart patterns</td>
<td valign="top" align="center">16.00</td>
</tr></tbody>
</table>
</table-wrap>
<p>Specifically, the Pact subset covers comprehensive question&#x02013;answering scenarios involving financial contracts. The Price subset encompasses nearly all complex candlestick-chart QA scenarios, including: (i) basic information comprehension: opening price, closing price, and intraday range; (ii) basic numerical reasoning: moving averages, relative volume (volume ratio), volatility, and percentage return; (iii) advanced computations: MACD, RSI, Williams %R (WR), and actual traded volume; (iv) technical analysis tactics: golden cross, death cross, and WR overbought/oversold signals; and (v) chart patterns: three black crows, ascending channel, descending channel, and the &#x0201C;air-refueling&#x0201D; (mid-trend consolidation) pattern.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Methodology</title>
<p>This section describes the training procedure, which is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. In this work we adopt a chain of thought paradigm because financial decision-making typically involves multi-step reasoning rather than one-shot classification. Supervising the model to generate intermediate rationales encourages it to decompose each task into economically meaningful steps instead of relying on shallow correlations. Building on this idea, the two stages framework first uses Supervised-CoT-Learning to imprint domain-faithful reasoning patterns from human annotated chains, and then applies Self-CoT-Learning to expand and stabilize this behavior on a larger set of examples without additional expert labeling. This design not only yields higher predictive accuracy in experiments, but also produces explanations that can be inspected by practitioners and regulators, making the resulting system better aligned with the transparency and accountability requirements of robo-advisory and risk management.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>The pipeline for FinErva fine-tuned model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1752580-g0002.tif">
<alt-text content-type="machine-generated">Flowchart titled &#x0201C;FinErva Fine-Tuned Interaction&#x0201D; depicting a conversation about identifying a bullish crossover in a stock chart. The left side shows a person asking when the magenta moving average crosses above the green line. The chart below shows a bullish crossover on October 26. Arrows lead from the person to two bots: &#x0201C;Supervised CoT&#x0201D; explaining the lines&#x02019; interaction and &#x0201C;Self CoT&#x0201D; confirming the crossover around October 26.</alt-text>
</graphic>
</fig>
<sec>
<label>4.1</label>
<title>Task definition</title>
<p>From a financial decision-support perspective, each instance in FinErva is designed to mimic a concrete advisory or analysis step, such as interpreting a contract clause, assessing a disclosure, or reading a candlestick pattern before making a trading or allocation decision. Formally, let <inline-formula><mml:math id="M1"><mml:mrow><mml:mi mathvariant="script">I</mml:mi></mml:mrow></mml:math></inline-formula> denote the space of visual inputs (e.g., <inline-formula><mml:math id="M2"><mml:mrow><mml:mi mathvariant="script">M</mml:mi></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow></mml:math></inline-formula> RGB images of contracts, screenshots, or candlestick charts), <inline-formula><mml:math id="M3"><mml:mrow><mml:mi mathvariant="script">Q</mml:mi></mml:mrow></mml:math></inline-formula> the discrete sequence space of natural-language questions posed in a financial context, <inline-formula><mml:math id="M4"><mml:mrow><mml:mi mathvariant="script">R</mml:mi></mml:mrow></mml:math></inline-formula> the space of chain of thought reasoning sequences, and <inline-formula><mml:math id="M5"><mml:mrow><mml:mi mathvariant="script">A</mml:mi></mml:mrow></mml:math></inline-formula> the space of answer sequences (multiple choice decisions).</p>
<p>A single sample in the multimodal financial question answering task can thus be formalized as a quadruple:</p>
<disp-formula id="EQ1"><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>x</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">vision</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">lang</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">I</mml:mi></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mi mathvariant="script">Q</mml:mi></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mi mathvariant="script">R</mml:mi></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mi mathvariant="script">A</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(1)</label></disp-formula>
<p>where <italic>x</italic><sub>vision</sub> is the visual input (such as a candlestick chart or contractual page), <italic>x</italic><sub>lang</sub> is the corresponding financial question (e.g., about fees, risk, or price behavior), <italic>r</italic><sup>&#x022C6;</sup> is an expert-style intermediate reasoning process that articulates economically and legally meaningful steps, and <italic>a</italic><sup>&#x022C6;</sup> is the final decision or answer. Intuitively, (<italic>r</italic><sup>&#x022C6;</sup>, <italic>a</italic><sup>&#x022C6;</sup>) plays the role of a &#x0201C;transparent advisory decision,&#x0201D; making explicit how an informed analyst would move from raw information to an actionable conclusion.</p>
</sec>
<sec>
<label>4.2</label>
<title>Target mapping</title>
<p>The goal is to learn a parameterized model <italic>F</italic><sub>&#x003B8;</sub> that takes the two modalities as input and outputs both a reasoning chain and a final decision:</p>
<disp-formula id="EQ2"><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mo>:</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="script">I</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="script">Q</mml:mi></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02192;</mml:mo><mml:mrow><mml:mi mathvariant="script">R</mml:mi></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mi mathvariant="script">A</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(2)</label></disp-formula>
<p>In financial terms, <italic>F</italic><sub>&#x003B8;</sub> can be viewed as an approximate decision policy that maps a given disclosure or market snapshot together with a user query into an interpretable rationale and a recommendation. Learning <italic>F</italic><sub>&#x003B8;</sub> therefore aims not only at predicting the correct answer, but also at recovering a step-by-step inference that can be scrutinized by practitioners, risk managers and regulators, in line with transparency and suitability expectations in robo-advisory.</p>
</sec>
<sec>
<label>4.3</label>
<title>Training target</title>
<p>The annotated dataset is denoted by <inline-formula><mml:math id="M8"><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow></mml:math></inline-formula>, where each training instance consists of a real visual input <italic>x</italic><sub>vision</sub>, a natural language question <italic>x</italic><sub>lang</sub>, an expert-annotated reasoning chain <italic>r</italic><sup>&#x022C6;</sup>, and a ground-truth answer <italic>a</italic><sup>&#x022C6;</sup>. The training objective <inline-formula><mml:math id="M9"><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is to minimize the negative log-likelihood of the pair (<italic>r</italic><sup>&#x022C6;</sup>, <italic>a</italic><sup>&#x022C6;</sup>):</p>
<disp-formula id="EQ3"><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">vision</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">lang</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0007E;</mml:mo><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mo class="qopname">log</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">vision</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">lang</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(3)</label></disp-formula>
<p>where</p>
<disp-formula id="E4"><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">vision</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">lang</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">vision</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">lang</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="EQ5"><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>&#x000B7;</mml:mo></mml:mtd><mml:mtd><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">vision</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">lang</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0003C;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(4)</label></disp-formula>
<p>and <xref ref-type="disp-formula" rid="EQ3">Equation 3</xref> is equivalent to maximizing <italic>p</italic><sub>&#x003B8;</sub>(<italic>r, a</italic>&#x02223;<italic>x</italic><sub>vision</sub>, <italic>x</italic><sub>lang</sub>), the joint conditional likelihood of the data. <italic>r</italic> &#x0003D; (<italic>s</italic><sub>1</sub>, &#x02026;, <italic>s</italic><sub><italic>T</italic><sub><italic>R</italic></sub></sub>) denote the tokenized reasoning sequence with length <italic>T</italic><sub><italic>R</italic></sub>, and <italic>a</italic> &#x0003D; (<italic>y</italic><sub>1</sub>, &#x02026;, <italic>y</italic><sub><italic>T</italic><sub><italic>A</italic></sub></sub>) denote the tokenized answer sequence with length <italic>T</italic><sub><italic>A</italic></sub>.</p>
<p>Economically, this objective encourages the model to learn not only which answers are correct, but also which <italic>reasoning patterns</italic> are consistent with expert financial practice: for instance, checking key cost and risk disclosures before judging product suitability, or combining trend, volatility and support/resistance levels before characterizing a candlestick configuration. Maximizing the joint likelihood <italic>p</italic><sub>&#x003B8;</sub>(<italic>r, a</italic>&#x02223;<italic>x</italic><sub>vision</sub>, <italic>x</italic><sub>lang</sub>) therefore aligns the model with the dual goal of modern advisory systems: accurate decisions and traceable, domain-consistent justifications.</p>
</sec>
<sec>
<label>4.4</label>
<title>Two stages optimization</title>
<p>Formulation (<xref ref-type="disp-formula" rid="EQ5">Equation 4</xref>) reflects an auto-regressive decomposition under the <italic>reasoning-then-answering</italic> paradigm: the model first generates a complete reasoning chain <italic>r</italic> &#x02208; <inline-formula><mml:math id="M13"><mml:mi mathvariant="script">R</mml:mi></mml:math></inline-formula><sup>&#x022C6;</sup>, conditioned on the input pair (<italic>x</italic><sub>vision</sub>, <italic>x</italic><sub>lang</sub>), and subsequently generates the final answer <italic>a</italic> &#x02208; <inline-formula><mml:math id="M14"><mml:mi mathvariant="script">A</mml:mi></mml:math></inline-formula><sup>&#x022C6;</sup> based on both the input and the generated reasoning chain.</p>
<sec>
<label>4.4.1</label>
<title>Vision encoding</title>
<p>Since most lightweight text generation models do not natively support multimodal inputs, visual features must be extracted first by using a Vision Transformer (ViT) (<xref ref-type="bibr" rid="B13">Dosovitskiy et al., 2021</xref>; <xref ref-type="bibr" rid="B45">Touvron et al., 2021</xref>). Specifically, a visual encoding layer is introduced by removing the classification head of the ViT. The encoded visual features, <inline-formula><mml:math id="M15"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">vision</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">I</mml:mi></mml:mrow><mml:mo>&#x02282;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, are then directly concatenated with the text embeddings and fed into the model during training, which significantly reduces overall training time.</p></sec>
<sec>
<label>4.4.2</label>
<title>Supervised-CoT-learning</title>
<p>In this stage, the training target is, given the input pair (<italic>x</italic><sub>vision</sub>, <italic>x</italic><sub>lang</sub>), to minimize:</p>
<disp-formula id="EQ6"><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">CoT</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">log</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">vision</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">lang</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0003C;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(5)</label></disp-formula>
<p>which denotes the model first generating a sequence of reasoning steps <italic>s</italic><sub>1</sub>, <italic>s</italic><sub>2</sub>, &#x02026;, <italic>s</italic><sub><italic>T</italic><sub><italic>R</italic></sub></sub>, conditioned on the given image and question. Each step <italic>s</italic><sub><italic>t</italic></sub> is generated based on the previously generated steps <italic>s</italic><sub>&#x0003C;<italic>t</italic></sub>. This encourages the model to reproduce the expert-annotated step-by-step reasoning sequence <italic>r</italic><sup>&#x022C6;</sup>.</p></sec>
<sec>
<label>4.4.3</label>
<title>Self-CoT-learning</title>
<p>In the second stage, once the model has generated its own reasoning chain <inline-formula><mml:math id="M17"><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow></mml:math></inline-formula>, we concatenate this chain with the original image&#x02013;question pair and use the resulting triplet as the input to this stage:</p>
<disp-formula id="E7"><mml:math id="M18"><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">lang</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">lang</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x02225;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>and the training target is to minimize:</p>
<disp-formula id="EQ8"><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Ans</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">log</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">vision</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo class="qopname">&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">lang</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0003C;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(6)</label></disp-formula>
<p>which denotes the model generating the answer tokens <italic>y</italic><sub><italic>k</italic></sub> after the complete reasoning chain <italic>r</italic>, conditioned on the inputs (<italic>x</italic><sub>vision</sub>, <italic>x</italic><sub>lang</sub>, <italic>r</italic>) and the previously generated answer tokens <italic>y</italic><sub>&#x0003C;<italic>k</italic></sub>. This encourages the model to generate more refined and accurate answers conditioned on its self-generated reasoning.</p>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Experiments</title>
<sec>
<label>5.1</label>
<title>Training details</title>
<sec>
<label>5.1.1</label>
<title>Lightweight models</title>
<p>This work focuses on training lightweight models, as their lower deployment cost offers significant advantages for future applications in personalized intelligent financial advisory systems. All models used in the experiments have fewer than 0.8 billion parameters, with the smallest model containing only 0.06 billion parameters, making it feasible to train on a single NVIDIA RTX 3090 GPU.</p></sec>
<sec>
<label>5.1.2</label>
<title>Training time</title>
<p>To further reduce training time, all experiments are conducted using two RTX 3090 GPUs in parallel, with each training run completed in under 8 hours. Small models only using about 2 hours. These statistics are provided solely to illustrate that the experiments adhere to the study&#x00027;s objective of maintaining low computational cost. They are based only on the local experimental environment and are intended for reference rather than for statistical inference.</p></sec>
<sec>
<label>5.1.3</label>
<title>Cross-validation</title>
<p>To prevent any potential data leakage among the training, validation and test portions of the 0.70/0.15/0.15 split, all reported performance metrics are obtained by averaging over five-fold cross-validation conducted exclusively within the combined 85% (train and val) non-test subset of the data. After selecting the optimal hyperparameters via cross-validation, the model is retrained on the entire 85% training pool and subsequently evaluated once on the strictly held-out 15% test set to obtain the final reported results.</p></sec>
<sec>
<label>5.1.4</label>
<title>Hyperparameter selection</title>
<p>To guarantee experimental reproducibility, we set a global random seed at 42. For the remaining hyperparameter choices, the settings may vary across different training devices and environments. Taking batch size as an example, an excessively large batch size may lead to out-of-memory issues, whereas an overly small batch size may result in under utilization of computational resources. The appropriate choice depends on the specific experimental setup. The selection of hyperparameter can cause slight fluctuations in training results, which is widely acknowledged in practice.</p>
</sec>
</sec>
<sec>
<label>5.2</label>
<title>Evaluation</title>
<p>All models&#x00027; performance will be assessed from two complementary perspectives in three different training stages, which are reported in <xref ref-type="table" rid="T7">Table 7</xref> respectively. First, zero-shot evaluation on test set; second, in the two-stage chain of thought pipeline without fine-tuning; third, after fine-tuning on FinErva dataset.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>FinErva-Pact results.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th/>
<th valign="top" align="center"><bold>Google-flan-T5-small</bold></th>
<th valign="top" align="center"><bold>Google-flan-T5-base</bold></th>
<th valign="top" align="center"><bold>Google-flan-T5-large</bold></th>
<th valign="top" align="center"><bold>Lamini-flan-T5-77M</bold></th>
<th valign="top" align="center"><bold>Lamini-flan-T5-248M</bold></th>
<th valign="top" align="center"><bold>Lamini-flan-T5-783M</bold></th>
<th valign="top" align="center"><bold>Flan-alpaca-base</bold></th>
<th valign="top" align="center"><bold>Flan-alpaca-large</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Parameters</td>
<td valign="top" align="center">60M</td>
<td valign="top" align="center">220M</td>
<td valign="top" align="center">770M</td>
<td valign="top" align="center">77M</td>
<td valign="top" align="center">248M</td>
<td valign="top" align="center">783M</td>
<td valign="top" align="center">220M</td>
<td valign="top" align="center">770M</td>
</tr>
<tr>
<td valign="top" align="left">Accuracy/%</td>
<td valign="top" align="center">20.77</td>
<td valign="top" align="center">20.59</td>
<td valign="top" align="center">20.59</td>
<td valign="top" align="center">20.59</td>
<td valign="top" align="center">20.59</td>
<td valign="top" align="center">20.59</td>
<td valign="top" align="center">20.59</td>
<td valign="top" align="center">20.59</td>
</tr>
<tr>
<td valign="top" align="left">ROUGE-1/%</td>
<td valign="top" align="center">33.61</td>
<td valign="top" align="center">38.95</td>
<td valign="top" align="center">38.30</td>
<td valign="top" align="center">33.94</td>
<td valign="top" align="center">43.65</td>
<td valign="top" align="center">45.64</td>
<td valign="top" align="center">38.81</td>
<td valign="top" align="center">44.07</td>
</tr>
<tr>
<td valign="top" align="left">ROUGE-2/%</td>
<td valign="top" align="center">11.40</td>
<td valign="top" align="center">14.47</td>
<td valign="top" align="center">15.29</td>
<td valign="top" align="center">11.15</td>
<td valign="top" align="center">16.62</td>
<td valign="top" align="center">17.99</td>
<td valign="top" align="center">15.15</td>
<td valign="top" align="center">17.15</td>
</tr>
<tr>
<td valign="top" align="left">ROUGE-L/%</td>
<td valign="top" align="center">27.97</td>
<td valign="top" align="center">30.41</td>
<td valign="top" align="center">30.86</td>
<td valign="top" align="center">26.63</td>
<td valign="top" align="center">33.42</td>
<td valign="top" align="center">34.97</td>
<td valign="top" align="center">30.48</td>
<td valign="top" align="center">34.29</td>
</tr>
<tr>
<td valign="top" align="left">Similarity/%</td>
<td valign="top" align="center">59.33</td>
<td valign="top" align="center">66.35</td>
<td valign="top" align="center">67.01</td>
<td valign="top" align="center">59.46</td>
<td valign="top" align="center">70.32</td>
<td valign="top" align="center">72.03</td>
<td valign="top" align="center">67.36</td>
<td valign="top" align="center">71.69</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Each evaluation has three lines, which indicates models&#x00027; performance under three different evaluation stages: first, zero-shot evaluation on test set; second, in the two-stage chain of thought pipeline without fine-tuning; third, after fine-tuning on FinErva dataset. The red color values indicate the best performance model.</p>
</table-wrap-foot>
</table-wrap>
<sec>
<label>5.2.1</label>
<title>Accuracy</title>
<p>Because each question has exactly one correct choice, Accuracy directly reflects decision reliability. <inline-formula><mml:math id="M20"><mml:mrow><mml:mrow><mml:mi mathvariant="script">T</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula> denotes the test space, <inline-formula><mml:math id="M21"><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x000E2;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">A</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula> the predicted option for sample <italic>i</italic>, and <inline-formula><mml:math id="M22"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> the ground truth. Accuracy is calculated as following:</p>
<disp-formula id="EQ9"><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi mathvariant="script">T</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi mathvariant="script">T</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:munderover></mml:mstyle><mml:mi>1</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x000E2;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(7)</label></disp-formula></sec>
<sec>
<label>5.2.2</label>
<title>ROUGE</title>
<p>Using ROUGE (Recall-Oriented Understudy for Gisting Evaluation) (<xref ref-type="bibr" rid="B26">Lin, 2004</xref>) score to quantify the solution <inline-formula><mml:math id="M24"><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow></mml:math></inline-formula> generated by model under given <italic>r</italic><sup>&#x022C6;</sup>, and computing Recall and Precision:</p>
<disp-formula id="E10"><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>g</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:munder></mml:mstyle><mml:mo class="qopname">min</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Count</mml:mtext></mml:mrow><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Count</mml:mtext></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>g</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Count</mml:mtext></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="EQ11"><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>g</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:munder></mml:mstyle><mml:mo class="qopname">min</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Count</mml:mtext></mml:mrow><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Count</mml:mtext></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>g</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Count</mml:mtext></mml:mrow><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(8)</label></disp-formula>
<p>where <italic>g</italic> spans all reference <italic>N</italic>-grams (with <italic>N</italic>&#x02208;{1, 2}), and ROUGE-L is computed via the longest common subsequence (LCS). <bold>ROUGE-1</bold>, <bold>ROUGE-2</bold>, and <bold>ROUGE-L</bold> are reported in results. <xref ref-type="table" rid="T7">Tables 7</xref>, <xref ref-type="table" rid="T8">8</xref> report the harmonic mean of recall (<italic>R</italic>) and precision (<italic>P</italic>), computed as</p>
<disp-formula id="E12"><mml:math id="M27"><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x000B7;</mml:mo><mml:mi>R</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>P</mml:mi></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>FinErva-Price results.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th/>
<th valign="top" align="center"><bold>Google-flan-T5-small</bold></th>
<th valign="top" align="center"><bold>Google-flan-T5-base</bold></th>
<th valign="top" align="center"><bold>Google-flan-T5-large</bold></th>
<th valign="top" align="center"><bold>Lamini-flan-T5-77M</bold></th>
<th valign="top" align="center"><bold>Lamini-flan-T5-248M</bold></th>
<th valign="top" align="center"><bold>Lamini-flan-T5-783M</bold></th>
<th valign="top" align="center"><bold>Flan-alpaca-base</bold></th>
<th valign="top" align="center"><bold>Flan-alpaca-large</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Parameters</td>
<td valign="top" align="center">60M</td>
<td valign="top" align="center">220M</td>
<td valign="top" align="center">770M</td>
<td valign="top" align="center">77M</td>
<td valign="top" align="center">248M</td>
<td valign="top" align="center">783M</td>
<td valign="top" align="center">220M</td>
<td valign="top" align="center">770M</td>
</tr>
<tr>
<td valign="top" align="left">Accuracy/%</td>
<td valign="top" align="center">22.61</td>
<td valign="top" align="center">18.88</td>
<td valign="top" align="center">18.88</td>
<td valign="top" align="center">22.61</td>
<td valign="top" align="center">22.61</td>
<td valign="top" align="center">18.88</td>
<td valign="top" align="center">18.88</td>
<td valign="top" align="center">18.88</td>
</tr>
<tr>
<td valign="top" align="left">ROUGE-1/%</td>
<td valign="top" align="center">12.24</td>
<td valign="top" align="center">24.32</td>
<td valign="top" align="center">32.56</td>
<td valign="top" align="center">10.85</td>
<td valign="top" align="center">23.38</td>
<td valign="top" align="center">26.90</td>
<td valign="top" align="center">26.31</td>
<td valign="top" align="center">32.56</td>
</tr>
<tr>
<td valign="top" align="left">ROUGE-2/%</td>
<td valign="top" align="center">1.04</td>
<td valign="top" align="center">8.22</td>
<td valign="top" align="center">11.33</td>
<td valign="top" align="center">1.74</td>
<td valign="top" align="center">7.19</td>
<td valign="top" align="center">9.81</td>
<td valign="top" align="center">8.74</td>
<td valign="top" align="center">11.33</td>
</tr>
<tr>
<td valign="top" align="left">ROUGE-L/%</td>
<td valign="top" align="center">10.92</td>
<td valign="top" align="center">20.25</td>
<td valign="top" align="center">24.58</td>
<td valign="top" align="center">9.22</td>
<td valign="top" align="center">16.96</td>
<td valign="top" align="center">19.91</td>
<td valign="top" align="center">20.31</td>
<td valign="top" align="center">24.58</td>
</tr>
<tr>
<td valign="top" align="left">Similarity/%</td>
<td valign="top" align="center">23.54</td>
<td valign="top" align="center">53.10</td>
<td valign="top" align="center">62.83</td>
<td valign="top" align="center">27.40</td>
<td valign="top" align="center">57.52</td>
<td valign="top" align="center">60.32</td>
<td valign="top" align="center">56.53</td>
<td valign="top" align="center">62.83</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Each evaluation has three lines, which indicates models&#x00027; performance under three different evaluation stages: first, zero-shot evaluation on test set; second, in the two-stage chain of thought pipeline without fine-tuning; third, after fine-tuning on FinErva dataset. The red color values indicate the best performance model.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<label>5.2.3</label>
<title>Similarity</title>
<p>The similarity score is obtained by first encoding both the generated and the annotated chains of thought into embeddings (as <bold>e</bold><sub><italic>a</italic></sub>, <bold>e</bold><sub><italic>b</italic></sub>) using the Sentence-BERT (<xref ref-type="bibr" rid="B36">Reimers and Gurevych, 2019</xref>), and then computing their cosine similarity. In experiments, we adopt the lightweight all-MiniLM-L6-v2 Sentence-BERT model, which is specifically designed for sentence embedding and semantic similarity computation. This choice ensures the reliability of the experimental results while maintaining the lightweight nature of our approach. The computation is given by:</p>
<disp-formula id="EQ13"><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mo class="qopname">sim</mml:mo></mml:mrow><mml:mrow><mml:mo class="qopname">cos</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant='bold-italic'><mml:mtext>e</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant='bold-italic'><mml:mtext>e</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant='bold-italic'><mml:mtext>e</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msubsup><mml:msub><mml:mrow><mml:mstyle mathvariant='bold-italic'><mml:mtext>e</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>&#x02225;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant='bold-italic'><mml:mtext>e</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x02225;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02225;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant='bold-italic'><mml:mtext>e</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x02225;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(9)</label></disp-formula>
<p>This metric reflects the semantic relatedness between the generated chain of thought and the human-annotated chain of thought.</p>
</sec>
</sec>
<sec>
<label>5.3</label>
<title>Results and analysis</title>
<p>In the financial domain, the output modality is almost exclusively textual, and the input modality is also predominantly either text-only or text&#x02013;vision. Accordingly, the author fine-tuned the classical text-to-text T5 (Text-to-Text Transfer Transformer) models, which have robust performance on text-vision2text task (<xref ref-type="bibr" rid="B34">Raffel et al., 2020</xref>; <xref ref-type="bibr" rid="B48">Wei et al., 2021</xref>). All models selected in experiments are lightweight, and results demonstrate that these lightweight models still exhibit strong capabilities in financial question answering tasks. The evaluated models include the <monospace>google/Flan-T5</monospace> family (Chung et al., <xref ref-type="bibr" rid="B9">2022</xref>), as well as instruction-tuned variants such as the <monospace>LaMini-Flan-T5</monospace> family (<xref ref-type="bibr" rid="B50">Wu et al., 2023</xref>) and the <monospace>alpaca-flan</monospace> family (<xref ref-type="bibr" rid="B2">Bhardwaj and Poria, 2023</xref>).</p>
<sec>
<label>5.3.1</label>
<title>Main results</title>
<p>The test results of the eight evaluated models on the two subsets are presented in <xref ref-type="table" rid="T7">Tables 7</xref>, <xref ref-type="table" rid="T8">8</xref>. In the zero-shot evaluation, all models achieve comparable accuracy on both subsets. However, in the two-stage evaluation without fine-tuning, accuracy improves significantly across all models, demonstrating the pronounced effect of chain-of-thought reasoning in multimodal financial QA tasks. Overall, there is a positive correlation between model size and accuracy, which is consistent with expectations. From a financial perspective, moving from low performance to substantially higher accuracy on FinErva-Pact translates into a lower probability that an automated system misinterprets key contractual or disclosure items, thereby reducing the risk of mis-selling and suitability breaches. Similarly, the gains observed on FinErva-Price indicate that the models more reliably recognize economically meaningful price configurations and basic risk signals in candlestick charts, which is a prerequisite for supporting trading discipline and avoiding systematically biased entry or exit decisions. Although experiments are conducted in an offline setting, these improvements in predictive accuracy can be interpreted as proxies for fewer interpretive errors and enhanced investor protection when such components are embedded into real-world robo-advisory workflows.</p></sec>
<sec>
<label>5.3.2</label>
<title>Higher learning cost in larger models</title>
<p>It is worth noting that in part of the FinErva-Price results, large-parameter models exhibit slightly lower answer accuracy after generating reasoning chains compared to medium-sized models (as shown in the second row of the Accuracy metric). A closer inspection of the intermediate reasoning chains reveals that the semantic richness and diversity of the CoTs generated by medium-sized models are noticeably higher than those of the large models.</p>
<p>This phenomenon can be explained by two main factors. First, the reasoning ability of different model series&#x02014;each fine-tuned under distinct instruction sets&#x02014;naturally varies in financial QA tasks. This is reflected in the first row of the Accuracy metric and is primarily determined by the pretraining stage, rather than the fine-tuning process or the dataset itself.</p>
<p>Second, larger models require significantly higher fine-tuning costs and larger training datasets. In the early stages of fine-tuning (the first iteration, as shown in the second accuracy row), large models demand a greater number of samples to achieve stable learning. As a result, they may initially underperform compared to medium-sized models. However, large models typically learn faster, and once the data scale reaches a sufficient threshold, they often enter a performance plateau, where further improvements become marginal. As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, when trained with only 50% of the full dataset, both small and medium models experience a noticeable drop in performance, whereas the large model&#x00027;s accuracy remains almost unchanged. This indicates that the small and medium models are still in the active learning phase, while the large model has already reached its saturation or plateau stage.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Comparison of model performance under different training set scales.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1752580-g0003.tif">
<alt-text content-type="machine-generated">Bar chart comparing model accuracy percentages on half versus full training sets for various models. Lamini-77M shows 57.5% on half and higher percentages on full training, while Google-770M achieves 80.9% on both datasets.</alt-text>
</graphic>
</fig>
<p>Such behavior is particularly evident in more complex reasoning tasks. In contrast, this phenomenon does not appear in FinErva-Pact, further confirming the high quality and stability of our dataset.</p></sec>
<sec>
<label>5.3.3</label>
<title>Limitations of ROUGE for CoT evaluation</title>
<p>Although the ROUGE metric is employed in this work to evaluate the semantic similarity of reasoning chains, the observed improvements in ROUGE scores before and after fine-tuning are relatively limited. This indicates the inherent limitations of ROUGE when applied to reasoning-based or semantically rich tasks. As a lexical overlap&#x02013;based metric, ROUGE primarily focuses on surface-level token or phrase matching, without genuinely capturing the semantic meaning or logical structure of the reasoning process. Moreover, the metric is highly sensitive to the annotators&#x00027; linguistic and cognitive styles, which may differ substantially from the instructional patterns learned by large pre-trained models. Therefore, while ROUGE can provide a coarse quantitative reference, it does not fully reflect the semantic or reasoning-level alignment between the model-generated and human-annotated CoTs.</p></sec>
<sec>
<label>5.3.4</label>
<title>Initial accuracy lower than random-guessing</title>
<p>As results in the first line of Accuracy, the fact that the initial accuracy falls below the random-guessing (0.33) baseline itself attests to the intrinsic complexity of this dataset. Because the questions incorporate substantial visual information, the model&#x00027;s predictions are easily misled, preventing it from extracting the correct answer from such rich visual content. Indeed, an initial accuracy lower than the mathematical expectation of random selection precisely indicates that the model is striving to interpret both complex visual and textual signals&#x02014;and in doing so, it further highlights the dataset&#x00027;s challenging nature.</p></sec>
<sec>
<label>5.3.5</label>
<title>Not perfect similarity score</title>
<p>Similarity measure precisely captures the semantic relationship between the generated and annotated chains of thought. Results in <xref ref-type="table" rid="T7">Tables 7</xref>, <xref ref-type="table" rid="T8">8</xref> demonstrate that the chains learned by the large model are semantically similar to the human-annotated chains, even though the numeric similarity scores are not particularly high. This is because cosine similarity reflects only the structural resemblance between two texts and does not truly capture their semantic content. To further illustrate this phenomenon, we conducted an additional experiment using the following three sentences:</p>
<p>sent-A = &#x0201C;I attended a meeting at the office this morning.&#x0201D;</p>
<p>sent-B = &#x0201C;I just wrapped up an early-morning business discussion.&#x0201D;</p>
<p>sent-C = &#x0201C;I had breakfast at the office this morning.&#x0201D;</p>
<p>The computed semantic similarity by all-MiniLM-L6-v2 between A and B is only 0.47, despite the two sentences expressing nearly identical meanings. In contrast, the similarity between A and C reaches 0.63, even though their semantic content is entirely unrelated. This suggests that current Sentence-BERT predominantly capture surface-level lexical or syntactic resemblance rather than genuine semantic equivalence. Consequently, quantitative semantic similarity metrics exhibit inherent limitations, which explains why the similarity evaluation indicators in our experiments are not perfectly aligned with true semantic consistency.</p></sec>
<sec>
<label>5.3.6</label>
<title>Thinking inertia</title>
<p>Interestingly, an experiment shows that when the correct answer is always placed at option A, model tends to learn this latent pattern. On the test set, this setup leads to a 1&#x02013;2 percentage point increase in accuracy compared to the shuffled version. It means that, if all correct answers are always presented in the same position, the large model may learn this superficial pattern. However, as the experimental results reveal, the impact of this positional bias is negligible.</p>
</sec>
</sec>
</sec>
<sec sec-type="conclusions" id="s6">
<label>6</label>
<title>Conclusion</title>
<p>This paper proposes FinErva as a new building block for data-driven, yet interpretable, financial decision support. From a financial perspective, the framework responds to three structural needs that arise in modern investment practice: (1) the ability to reason jointly over heterogeneous information sources such as contracts, disclosures, market scenes and candlestick charts; (2) the requirement that automated advice be transparent enough to withstand scrutiny from investors, risk managers and regulators; and (3) the necessity of keeping modeling and deployment costs at a level that is feasible for financial institutions beyond a small set of frontier AI labs.</p>
<p>FinErva is, to our knowledge, the first multimodal chain-of-thought dataset specifically designed for financial reasoning. It integrates real-world financial contracts and candlestick charts with fine-grained reasoning annotations, yielding 7,544 manually validated samples across two complementary subsets: FinErva-Pact for contract and disclosure understanding, and FinErva-Price for price-pattern and technical-analysis reasoning. Each instance includes multimodal inputs and a human-supervised reasoning chain, which enables joint evaluation of answer accuracy and the quality of the underlying rationale rather than focusing solely on black-box predictive performance.</p>
<p>On top of this dataset, we design and empirically validate a two-stage training paradigm: Supervised-CoT Learning followed by Self-CoT Refinement, for lightweight vision&#x02013;language models with fewer than 0.8 billion parameters. The results show that explicit reasoning supervision substantially improves performance over zero-shot and standard fine-tuning baselines and allows lightweight models to approach the reasoning competence of domain experts while being trainable on commodity hardware. For practitioners, this suggests that expert-level multimodal reasoning for tasks such as contract review, chart-based signal extraction and scenario analysis does not necessarily require frontier models, but can be achieved through targeted, domain aligned CoT adaptation.</p>
<p>Beyond serving as a standalone framework, FinErva highlights the broader role of structured reasoning supervision in the design of trustworthy financial AI. The findings indicate that the path toward robust multimodal robo-advisory and risk-management systems lies not only in scaling models, but also in aligning their intermediate reasoning processes with the interpret ability and audit ability requirements of financial economics.</p>
<p>This work has several limitations that open avenues for future research. The current dataset focuses on English-language materials and a subset of visual artifacts (contracts and candlestick charts); extending FinErva to tabular and time-series data, additional document types and multilingual, cross-market settings would strengthen its coverage of global financial practice. From an ethical perspective, the dataset may contain biases, face limitations in its generalizability across jurisdictions and market structures, and raise questions about how to enhance decision-making capabilities while simultaneously addressing ethical and governance concerns. Moreover, integrating FinErva into open evaluation frameworks for retrieval-augmented and reinforcement-tuned reasoning systems would facilitate systematic comparison of alternative architectures for interpretable financial AI. Taken together, these directions position FinErva as a foundation for the next generation of personalized, explainable and operationally viable multimodal intelligence in finance.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The source code of the project is available at [<ext-link ext-link-type="uri" xlink:href="https://github.com/JerryChi222/FinErva-Interpretable-Multimodal-Reasoning-for-Robo-Advisory.git">https://github.com/JerryChi222/FinErva-Interpretable-Multimodal-Reasoning-for-Robo-Advisory.git</ext-link>], and the dataset can be accessed at [<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/datasets/jerrychi/FinErva">https://huggingface.co/datasets/jerrychi/FinErva</ext-link>].</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>JC: Software, Supervision, Methodology, Funding acquisition, Writing &#x02013; original draft, Conceptualization, Investigation, Writing &#x02013; review &#x00026; editing, Visualization, Formal analysis, Resources, Validation, Project administration, Data curation.</p>
</sec>
<ack><title>Acknowledgments</title><p>The author gratefully acknowledge the scholars who offered constructive guidance to this work but are not listed as co-authors.</p></ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The author(s) declared that this work was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declared that generative AI was not used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bagnara</surname> <given-names>M.</given-names></name></person-group> (<year>2024</year>). <article-title>Asset pricing and machine learning: a critical review</article-title>. <source>J. Econ. Surv</source>. <volume>38</volume>, <fpage>27</fpage>&#x02013;<lpage>56</lpage>. doi: <pub-id pub-id-type="doi">10.1111/joes.12532</pub-id></mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bhardwaj</surname> <given-names>R.</given-names></name> <name><surname>Poria</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>Red-teaming large language models using chain of utterances for safety-alignment</article-title>. <source>arXiv preprint arXiv:2308.09662</source>.</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boreiko</surname> <given-names>D.</given-names></name> <name><surname>Massarotti</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>How risk profiles of investors affect robo-advised portfolios</article-title>. <source>Front. Artif. Intell</source>. <volume>3</volume>:<fpage>60</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2020.00060</pub-id><pub-id pub-id-type="pmid">33733177</pub-id></mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bussmann</surname> <given-names>N.</given-names></name> <name><surname>Giudici</surname> <given-names>P.</given-names></name> <name><surname>Marinelli</surname> <given-names>D.</given-names></name> <name><surname>Papenbrock</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Explainable AI in fintech risk management</article-title>. <source>Front. Artif. Intell</source>. <volume>3</volume>:<fpage>26</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2020.00026</pub-id><pub-id pub-id-type="pmid">33733145</pub-id></mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chellappa</surname> <given-names>R.</given-names></name> <name><surname>Pramanick</surname> <given-names>S.</given-names></name> <name><surname>Venugopalan</surname> <given-names>S.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;SPIQA: a dataset for multimodal question answering on scientific papers,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, 118807&#x02013;118833. doi: <pub-id pub-id-type="doi">10.52202/079017-3773</pub-id></mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>L.</given-names></name> <name><surname>Pelger</surname> <given-names>M.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name></person-group> (<year>2024</year>). <article-title>Deep learning in asset pricing</article-title>. <source>Manage. Sci</source>. <volume>70</volume>, <fpage>714</fpage>&#x02013;<lpage>750</lpage>. doi: <pub-id pub-id-type="doi">10.1287/mnsc.2023.4695</pub-id></mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Smiley</surname> <given-names>C.</given-names></name> <name><surname>Shah</surname> <given-names>S.</given-names></name> <name><surname>Borova</surname> <given-names>I.</given-names></name> <name><surname>Langdon</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2022a</year>). <article-title>&#x0201C;FinQA: a dataset of numerical reasoning over financial data,&#x0201D;</article-title> in <source>Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing</source>. doi: <pub-id pub-id-type="doi">10.18653/v1/2021.emnlp-main.300</pub-id></mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Smiley</surname> <given-names>C.</given-names></name> <name><surname>Ma</surname> <given-names>Z.</given-names></name> <name><surname>Shah</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>W. Y.</given-names></name></person-group> (<year>2022b</year>). <article-title>&#x0201C;ConvFinQA: exploring the chain of numerical reasoning in conversational finance question answering,&#x0201D;</article-title> in <source>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</source>. doi: <pub-id pub-id-type="doi">10.18653/v1/2022.emnlp-main.421</pub-id></mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chung</surname> <given-names>H. W.</given-names></name> <name><surname>Hou</surname> <given-names>L.</given-names></name> <name><surname>Longpre</surname> <given-names>S.</given-names></name> <name><surname>Zoph</surname> <given-names>B.</given-names></name> <name><surname>Tai</surname> <given-names>Y.</given-names></name> <name><surname>Fedus</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Scaling instruction-finetuned language models</article-title>. <source>J. Mach. Learn. Res.</source> <volume>25</volume>, <fpage>3381</fpage>&#x02013;<lpage>3433</lpage>. doi: <pub-id pub-id-type="doi">10.5555/3722577.3722647</pub-id></mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cottier</surname> <given-names>B.</given-names></name> <name><surname>Rahman</surname> <given-names>R.</given-names></name> <name><surname>Fattorini</surname> <given-names>L.</given-names></name> <name><surname>Maslej</surname> <given-names>N.</given-names></name> <name><surname>Besiroglu</surname> <given-names>T.</given-names></name> <name><surname>Owen</surname> <given-names>D.</given-names></name></person-group> (<year>2025</year>). <article-title>The rising costs of training frontier AI models</article-title>. <source>arXiv preprint arXiv:2405.21015</source>.</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>D&#x00027;Acunto</surname> <given-names>F.</given-names></name> <name><surname>Prabhala</surname> <given-names>N.</given-names></name> <name><surname>Rossi</surname> <given-names>A. G.</given-names></name></person-group> (<year>2019</year>). <article-title>The promises and pitfalls of robo-advising</article-title>. <source>Rev. Financ. Stud</source>. <volume>32</volume>, <fpage>1983</fpage>&#x02013;<lpage>2020</lpage>. doi: <pub-id pub-id-type="doi">10.1093/rfs/hhz014</pub-id></mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>Y.</given-names></name> <name><surname>Luo</surname> <given-names>S.</given-names></name> <name><surname>Chung</surname> <given-names>H.</given-names></name> <name><surname>Han</surname> <given-names>S. C.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;PDF-VQA: a new dataset for real-world VQA on PDF documents,&#x0201D;</article-title> in <source>Machine Learning and Knowledge Discovery in Databases: Applied Data Science and Demo Track</source>, 585&#x02013;601. doi: <pub-id pub-id-type="doi">10.1007/978-3-031-43427-3_35</pub-id></mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname> <given-names>A.</given-names></name> <name><surname>Beyer</surname> <given-names>L.</given-names></name> <name><surname>Kolesnikov</surname> <given-names>A.</given-names></name> <name><surname>Weissenborn</surname> <given-names>D.</given-names></name> <name><surname>Zhai</surname> <given-names>X.</given-names></name> <name><surname>Unterthiner</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;An image is worth 16 &#x000D7; 16 words: Transformers for image recognition at scale,&#x0201D;</article-title> in <source>International Conference on Learning Representations (ICLR)</source>.</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eskandarany</surname> <given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>Adoption of artificial intelligence and machine learning in banking systems: a qualitative survey of board of directors</article-title>. <source>Front. Artif. Intell</source>. <volume>7</volume>:<fpage>1440051</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2024.1440051</pub-id><pub-id pub-id-type="pmid">39664101</pub-id></mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname> <given-names>L.</given-names></name> <name><surname>Qi</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name></person-group> (<year>2025</year>). <article-title>The spillover effects of the &#x0201C;Binance Incident&#x0201D; on financial markets: a study based on machine learning approach</article-title>. <source>Finance Res. Lett</source>. <volume>71</volume>:<fpage>106383</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.frl.2024.106383</pub-id></mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Flowers</surname> <given-names>J. G.</given-names></name></person-group> (<year>2025</year>). Finance-instruct-500k.</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fritz-Morgenthal</surname> <given-names>S.</given-names></name> <name><surname>Hein</surname> <given-names>B.</given-names></name> <name><surname>Papenbrock</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Financial risk management and explainable, trustworthy, responsible AI</article-title>. <source>Front. Artif. Intell</source>. <volume>5</volume>:<fpage>779799</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2022.779799</pub-id><pub-id pub-id-type="pmid">35295866</pub-id></mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gan</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>D.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>&#x0201C;MME-Finance: a multimodal finance benchmark for expert-level understanding and reasoning,&#x0201D;</article-title> in <source>Proceedings of the 33rd ACM International Conference on Multimedia</source>, 12867&#x02013;12874. doi: <pub-id pub-id-type="doi">10.1145/3746027.3758230</pub-id></mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Goswami</surname> <given-names>D.</given-names></name> <name><surname>Verma</surname> <given-names>B.</given-names></name> <name><surname>Kumar Sinha</surname> <given-names>S.</given-names></name> <name><surname>Mittal</surname> <given-names>A.</given-names></name></person-group> (<year>2025</year>). <article-title>Cracking the code of initial trust: pathways to adoption of financial robo-advisors via cognitive absorption</article-title>. <source>J. Internet Commerce</source> <volume>24</volume>, <fpage>287</fpage>&#x02013;<lpage>324</lpage>. doi: <pub-id pub-id-type="doi">10.1080/15332861.2025.2546494</pub-id></mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hegde</surname> <given-names>S.</given-names></name> <name><surname>Fazli</surname> <given-names>P.</given-names></name> <name><surname>Seifi</surname> <given-names>H.</given-names></name></person-group> (<year>2025</year>). <article-title>ChartQA-X: generating explanations for charts</article-title>. <source>arXiv preprint arXiv:2504.13275</source>.</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Interpress</surname> <given-names>V.</given-names></name></person-group> (<year>2024</year>). <source>Artificial intelligence and machine learning in finance: Addressing complex problems and ESG applications</source>. Virtus Interpress.</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jadhav</surname> <given-names>A.</given-names></name> <name><surname>Mirza</surname> <given-names>V.</given-names></name></person-group> (<year>2025</year>). <article-title>Large language models in equity markets: applications, techniques, and insights</article-title>. <source>Front. Artif. Intell</source>. <volume>8</volume>:<fpage>1608365</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2025.1608365</pub-id><pub-id pub-id-type="pmid">40936657</pub-id></mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jung</surname> <given-names>D.</given-names></name> <name><surname>Dorner</surname> <given-names>V.</given-names></name> <name><surname>Weinhardt</surname> <given-names>C.</given-names></name> <name><surname>Pusmaz</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Designing a robo-advisor for risk-averse, low-budget consumers</article-title>. <source>Electr. Markets</source> <volume>28</volume>, <fpage>367</fpage>&#x02013;<lpage>380</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s12525-017-0279-9</pub-id></mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kwon</surname> <given-names>T. Y.</given-names></name></person-group> (<year>2025</year>). <article-title>Feature importance in linear models with ensemble machine learning: a study of the Fama and French five-factor model</article-title>. <source>Finance Res. Lett</source>. <volume>71</volume>:<fpage>106406</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.frl.2024.106406</pub-id></mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Leong</surname> <given-names>J.</given-names></name> <name><surname>Di</surname> <given-names>K.</given-names></name> <name><surname>Cham</surname> <given-names>B.</given-names></name> <name><surname>Heng</surname> <given-names>S.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;Shdocs: a dataset, benchmark, and method to efficiently generate high-quality, real-world specular highlight data with near-perfect alignment,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source> (<publisher-loc>Curran Associates, Inc.</publisher-loc>).</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>C.-Y.</given-names></name></person-group> (<year>2004</year>). <article-title>&#x0201C;ROUGE: a package for automatic evaluation of summaries,&#x0201D;</article-title> in <source>Text Summarization Branches Out</source> (<publisher-loc>Barcelona, Spain</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>74</fpage>&#x02013;<lpage>81</lpage>.</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>F.</given-names></name> <name><surname>Song</surname> <given-names>Y.</given-names></name></person-group> (<year>2025</year>). <article-title>Analysis of credit ABS based on Markov chain approaches</article-title>. <source>Finance Res. Lett</source>. <volume>71</volume>:<fpage>106432</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.frl.2024.106432</pub-id></mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Guo</surname> <given-names>X.</given-names></name> <name><surname>Lou</surname> <given-names>F.</given-names></name> <name><surname>Zeng</surname> <given-names>L.</given-names></name> <name><surname>Niu</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Fin-r1: a large language model for financial reasoning through reinforcement learning</article-title>. <source>arXiv preprint arXiv:2503.16252</source>.</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>D.</given-names></name> <name><surname>Wu</surname> <given-names>H.</given-names></name> <name><surname>Liang</surname> <given-names>J.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>He</surname> <given-names>Q.</given-names></name> <name><surname>Geng</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Bbt-fin: Comprehensive construction of chinese financial domain pre-trained language model, corpus and benchmark</article-title>. <source>arXiv preprint arXiv:2302.09432</source>.</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maier</surname> <given-names>T.</given-names></name> <name><surname>Menold</surname> <given-names>J.</given-names></name> <name><surname>McComb</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>The relationship between performance and trust in AI in E-finance</article-title>. <source>Front. Artif. Intell</source>. <volume>5</volume>:<fpage>891529</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2022.891529</pub-id><pub-id pub-id-type="pmid">35800065</pub-id></mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maple</surname> <given-names>C.</given-names></name> <name><surname>Sabuncuoglu</surname> <given-names>A.</given-names></name> <name><surname>Szpruch</surname> <given-names>L.</given-names></name> <name><surname>Elliott</surname> <given-names>A.</given-names></name> <name><surname>Reinert</surname> <given-names>T. Z. G.</given-names></name></person-group> (<year>2024</year>). <source>The impact of large language models in finance: Towards trustworthy adoption</source>. Technical report, The Alan Turing Institute.</mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paleyes</surname> <given-names>A.</given-names></name> <name><surname>Urma</surname> <given-names>R.-G.</given-names></name> <name><surname>Lawrence</surname> <given-names>N. D.</given-names></name></person-group> (<year>2022</year>). <article-title>Challenges in deploying machine learning: a survey of case studies</article-title>. <source>ACM Comput. Surv</source>. <volume>55</volume>, <fpage>1</fpage>&#x02013;<lpage>29</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3533378</pub-id></mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Qian</surname> <given-names>L.</given-names></name> <name><surname>Zhou</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Peng</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>J.</given-names></name> <name><surname>Xie</surname> <given-names>Q.</given-names></name></person-group> (<year>2025</year>). <article-title>Fino1: on the transferability of reasoning enhanced llms to finance</article-title>. <source>arXiv e-prints, arXiv-2502</source>.</mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Raffel</surname> <given-names>C.</given-names></name> <name><surname>Shazeer</surname> <given-names>N.</given-names></name> <name><surname>Roberts</surname> <given-names>A.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Narang</surname> <given-names>S.</given-names></name> <name><surname>Matena</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Exploring the limits of transfer learning with a unified text-to-text transformer</article-title>. <source>J. Mach. Learn. Res</source>. <volume>21</volume>, <fpage>1</fpage>&#x02013;<lpage>67</lpage>.</mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rangapur</surname> <given-names>A.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Jian</surname> <given-names>L.</given-names></name> <name><surname>Shu</surname> <given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;Fin-Fact: a benchmark dataset for multimodal financial fact-checking and explanation generation,&#x0201D;</article-title> in <source>Companion Proceedings of the ACM on Web Conference</source>, 785&#x02013;788. doi: <pub-id pub-id-type="doi">10.1145/3701716.3715292</pub-id></mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Reimers</surname> <given-names>N.</given-names></name> <name><surname>Gurevych</surname> <given-names>I.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Sentence-BERT: sentence embeddings using siamese BERT-networks,&#x0201D;</article-title> in <source>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</source>, 3980&#x02013;3990. doi: <pub-id pub-id-type="doi">10.18653/v1/D19-1410</pub-id></mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sen</surname> <given-names>J.</given-names></name> <name><surname>Sen</surname> <given-names>R.</given-names></name> <name><surname>Dutta</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Introductory chapter: machine learning in finance-emerging trends and challenges,&#x0201D;</article-title> in <source>Machine Learning</source> - <italic>Algorithms, Models and Applications</italic>. doi: <pub-id pub-id-type="doi">10.5772/intechopen.101120</pub-id></mixed-citation>
</ref>
<ref id="B38">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Shao</surname> <given-names>H.</given-names></name> <name><surname>Qian</surname> <given-names>S.</given-names></name> <name><surname>Xiao</surname> <given-names>H.</given-names></name> <name><surname>Song</surname> <given-names>G.</given-names></name> <name><surname>Zong</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>&#x0201C;Visual cot: advancing multi-modal language models with a comprehensive dataset and benchmark for chain-of-thought reasoning,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 37 (Datasets and Benchmarks Track)</source> (<publisher-loc>Curran Associates, Inc.</publisher-loc>).</mixed-citation>
</ref>
<ref id="B39">
<mixed-citation publication-type="web"><person-group person-group-type="author"><name><surname>Sujet</surname> <given-names>A. I.</given-names></name> <name><surname>Allaa Boutaleb</surname> <given-names>H. R.</given-names></name></person-group> (<year>2024</year>). <source>Sujet-finance-qa-vision-100k: A large-scale dataset for financial document vqa</source>. <ext-link ext-link-type="uri" xlink:href="https://huggingface.co/datasets/sujet-a">https://huggingface.co/datasets/sujet-a</ext-link></mixed-citation>
</ref>
<ref id="B40">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sutiene</surname> <given-names>K.</given-names></name> <name><surname>Schwendner</surname> <given-names>P.</given-names></name> <name><surname>Sipos</surname> <given-names>C.</given-names></name> <name><surname>Lorenzo</surname> <given-names>L.</given-names></name> <name><surname>Mirchev</surname> <given-names>M.</given-names></name> <name><surname>Lameski</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Enhancing portfolio management using artificial intelligence: literature review</article-title>. <source>Front. Artif. Intell</source>. <volume>7</volume>:<fpage>1371502</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2024.1371502</pub-id><pub-id pub-id-type="pmid">38650961</pub-id></mixed-citation>
</ref>
<ref id="B41">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>E. H. L.</given-names></name> <name><surname>Hamed</surname> <given-names>Y.</given-names></name> <name><surname>Daud</surname> <given-names>H.</given-names></name> <name><surname>Abdul Wahab</surname> <given-names>M. A. F.</given-names></name> <name><surname>Azhar</surname> <given-names>A. A. A.</given-names></name> <name><surname>Tan</surname> <given-names>S. Y.</given-names></name></person-group> (<year>2025</year>). <article-title>Profiling investor behavior in the Malaysian derivatives market using K-means clustering</article-title>. <source>Front. Artif. Intell</source>. <volume>8</volume>:<fpage>1640776</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2025.1640776</pub-id><pub-id pub-id-type="pmid">41041086</pub-id></mixed-citation>
</ref>
<ref id="B42">
<mixed-citation publication-type="web"><person-group person-group-type="author"><name><surname>Team</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <source>Financial evaluation dataset</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://github.com/alipay/financial_evaluation_dataset">https://github.com/alipay/financial_evaluation_dataset</ext-link> (Accessed March 18, 2024).</mixed-citation>
</ref>
<ref id="B43">
<mixed-citation publication-type="web"><person-group person-group-type="author"><name><surname>Team</surname> <given-names>D. D.</given-names></name></person-group> (<year>2023</year>). <source>Financeiq</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://github.com/Duxiaoman-DI/XuanYuan/tree/main/FinanceIQ">https://github.com/Duxiaoman-DI/XuanYuan/tree/main/FinanceIQ</ext-link> (Accessed March 18, 2024).</mixed-citation>
</ref>
<ref id="B44">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Theodorakopoulos</surname> <given-names>L.</given-names></name> <name><surname>Theodoropoulou</surname> <given-names>A.</given-names></name> <name><surname>Bakalis</surname> <given-names>A.</given-names></name></person-group> (<year>2025</year>). <article-title>Big data in financial risk management: evidence, advances, and open questions: a systematic review</article-title>. <source>Front. Artif. Intell</source>. <volume>8</volume>:<fpage>1658375</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2025.1658375</pub-id><pub-id pub-id-type="pmid">41104145</pub-id></mixed-citation>
</ref>
<ref id="B45">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Touvron</surname> <given-names>H.</given-names></name> <name><surname>Cord</surname> <given-names>M.</given-names></name> <name><surname>Douze</surname> <given-names>M.</given-names></name> <name><surname>Massa</surname> <given-names>F.</given-names></name> <name><surname>Sablayrolles</surname> <given-names>A.</given-names></name> <name><surname>J&#x000E9;gou</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Training data-efficient image transformers &#x00026;distillation through attention,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)</source>, <fpage>10302</fpage>&#x02013;<lpage>10312</lpage>.</mixed-citation>
</ref>
<ref id="B46">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Verma</surname> <given-names>B.</given-names></name> <name><surname>Schulze</surname> <given-names>M.</given-names></name> <name><surname>Goswami</surname> <given-names>D.</given-names></name> <name><surname>Upreti</surname> <given-names>K.</given-names></name></person-group> (<year>2025</year>). <article-title>Artificial intelligence attitudes and resistance to use robo-advisors: exploring investor reluctance toward cognitive financial systems</article-title>. <source>Front. Artif. Intell</source>. <volume>8</volume>:<fpage>1623534</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2025.1623534</pub-id><pub-id pub-id-type="pmid">41041081</pub-id></mixed-citation>
</ref>
<ref id="B47">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Soon</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>Finvis-GPT: a multimodal large language model for financial chart analysis</article-title>. <source>arXiv preprint arXiv:2308.01430</source>.</mixed-citation>
</ref>
<ref id="B48">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname> <given-names>J.</given-names></name> <name><surname>Bosma</surname> <given-names>M.</given-names></name> <name><surname>Zhao</surname> <given-names>V. Y.</given-names></name> <name><surname>Guu</surname> <given-names>K.</given-names></name> <name><surname>Yu</surname> <given-names>A. W.</given-names></name> <name><surname>Lester</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Finetuned language models are zero-shot learners</article-title>. <source>arXiv preprint arXiv:2109.01652</source>.</mixed-citation>
</ref>
<ref id="B49">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Schuurmans</surname> <given-names>D.</given-names></name> <name><surname>Bosma</surname> <given-names>M.</given-names></name> <name><surname>Xia</surname> <given-names>F.</given-names></name> <name><surname>Chi</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>&#x0201C;Chain-of-thought prompting elicits reasoning in large language models,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, <fpage>24824</fpage>&#x02013;<lpage>24837</lpage>.</mixed-citation>
</ref>
<ref id="B50">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>M.</given-names></name> <name><surname>Waheed</surname> <given-names>A.</given-names></name> <name><surname>Zhang</surname> <given-names>C.</given-names></name> <name><surname>Abdul-Mageed</surname> <given-names>M.</given-names></name> <name><surname>Aji</surname> <given-names>A. F.</given-names></name></person-group> (<year>2023</year>). <article-title>Lamini-LM: a diverse herd of distilled models from large-scale instructions</article-title>. <source>CoRR, abs/2304.14402</source>.</mixed-citation>
</ref>
<ref id="B51">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Xie</surname> <given-names>Q.</given-names></name> <name><surname>Han</surname> <given-names>W.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Xiang</surname> <given-names>R.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>He</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>&#x0201C;Finben: a holistic financial benchmark for large language models,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 37 (Datasets and Benchmarks Track)</source> (<publisher-loc>Curran Associates, Inc.</publisher-loc>).</mixed-citation>
</ref>
<ref id="B52">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xue</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Zhou</surname> <given-names>F.</given-names></name> <name><surname>Dai</surname> <given-names>Q.</given-names></name> <name><surname>Chu</surname> <given-names>Z.</given-names></name> <name><surname>Mei</surname> <given-names>H.</given-names></name></person-group> (<year>2025</year>). <article-title>Famma: a benchmark for financial domain multilingual multimodal question answering</article-title>. <source>arXiv preprint arXiv:2410.04526</source>.</mixed-citation>
</ref>
<ref id="B53">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>A.</given-names></name> <name><surname>Li</surname> <given-names>M.</given-names></name> <name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Karypis</surname> <given-names>G.</given-names></name> <name><surname>Smola</surname> <given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>Multimodal chain-of-thought reasoning in language models</article-title>. <source>arXiv preprint arXiv:2302.00923</source>.</mixed-citation>
</ref>
<ref id="B54">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>F.</given-names></name> <name><surname>Lei</surname> <given-names>W.</given-names></name> <name><surname>Huang</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Lv</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;TAT-QA: a question answering benchmark on a hybrid of tabular and textual content in Finance,&#x0201D;</article-title> in <source>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)</source>. doi: <pub-id pub-id-type="doi">10.18653/v1/2021.acl-long.254</pub-id></mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/591664/overview">Peter Schwendner</ext-link>, Zurich University of Applied Sciences, Switzerland</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1354259/overview">Norbert Hilber</ext-link>, ZHAW, Switzerland</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3264683/overview">Marco Bonelli</ext-link>, Ca&#x00027; Foscari University of Venice, Italy</p>
</fn>
</fn-group>
<fn-group>
<fn id="fn0003"><label>1</label><p>Minerva is a goddess of wisdom in Roman mythology.</p></fn>
</fn-group>
</back>
</article>