<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1640776</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Profiling investor behavior in the Malaysian derivatives market using K-means clustering</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Tan</surname>
<given-names>Eng Hao Louis</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3151649/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Hamed</surname>
<given-names>Yaman</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3089590/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Daud</surname>
<given-names>Hanita</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Abdul Wahab</surname>
<given-names>Mohd Amirul Faiz</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Azhar</surname>
<given-names>Ahmad Amirul Adlan</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tan</surname>
<given-names>Sieow Yeek</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Applied Sciences, Intelligent Asset Reliability Centre, Institute of Emerging Digital Technologies, Universiti Teknologi PETRONAS</institution>, <addr-line>Seri Iskandar</addr-line>, <country>Malaysia</country></aff>
<aff id="aff2"><sup>2</sup><institution>Bursa Malaysia Berhad</institution>, <addr-line>Kuala Lumpur</addr-line>, <country>Malaysia</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2552970/overview">Maria Chiara Caschera</ext-link>, National Research Council (CNR), Italy</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2568214/overview">Iryna Mihus</ext-link>, Scientific Center of Innovative Research, Estonia</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3145475/overview">Yuquan Liu</ext-link>, University of Science and Technology of China, China</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3146067/overview">Francka Sakti Lee</ext-link>, University of Bunda Mulia, Indonesia</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3146070/overview">Sudha Palaniappan</ext-link>, SSN College of Engineering, India</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3149397/overview">Musli Yanto</ext-link>, Universitas Putra Indonesia, Indonesia</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Yaman Hamed, <email>yaman.hamed@utp.edu.my</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1640776</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>25</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Tan, Hamed, Daud, Abdul Wahab, Azhar and Tan.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Tan, Hamed, Daud, Abdul Wahab, Azhar and Tan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>This study investigates the trading behaviors of Malaysian derivatives traders using a comprehensive dataset from Bursa Malaysia with K-means clustering, representing one of the first AI applications to derivatives market segmentation. The analysis encompassed over 11 million trade records for FCPO and FKLI derivatives from January to December 2022. Six key features were engineered to segment derivative traders: Total Number of Trades, Total Traded Amount, Overall Realized Profit, Average ROI, Maximum Account Vintage (trader experience in years), and Median Holding Days (typical position duration). Inverse Hyperbolic Sine transformation was applied to address extreme outliers, ensuring robust feature scaling. K-means clustering identified five distinct profiles: &#x201C;High-Frequency, High-Risk Derivative Traders with Consistent Losses,&#x201D; &#x201C;Conservative, Steady-Growth Derivative Trader,&#x201D; &#x201C;High-Frequency, High-Yield Derivative Traders,&#x201D; &#x201C;Conservative, Low-Yield Derivative Traders,&#x201D; and &#x201C;Cautious, Low-Activity Novice Derivative Traders.&#x201D; Decision tree classifiers validated these clusters through interpretable splitting conditions. These profiles enable targeted risk management strategies, personalized trading services, and evidence-based regulatory policies for derivatives markets and future research.</p>
</abstract>
<kwd-group>
<kwd>clustering</kwd>
<kwd>K-means</kwd>
<kwd>decision trees</kwd>
<kwd>trading behavior</kwd>
<kwd>derivatives</kwd>
<kwd>investors behavior</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="2"/>
<equation-count count="5"/>
<ref-count count="27"/>
<page-count count="13"/>
<word-count count="7387"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>AI in Finance</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<title>Introduction</title>
<p>Derivatives are financial instruments whose value derives from underlying assets like commodities, stocks, or indices, used for hedging risks, speculation, and portfolio enhancement. Malaysia&#x2019;s most actively traded derivatives include Futures Crude Palm Oil (FCPO), linked to crude palm oil prices, and Futures Kuala Lumpur Index (FKLI). FCPO is a commodity-based derivative linked to the price of crude palm oil, a significant export commodity for Malaysia, making it particularly attractive for participants in the agricultural and commodities sectors (<xref ref-type="bibr" rid="ref15">Rizal et al., 2023</xref>; <xref ref-type="bibr" rid="ref10">Jamak, 2018</xref>). FKLI, is an index-based derivative that tracks the performance of the Bursa Malaysia Kuala Lumpur Composite Index, which represents the Malaysian stock market&#x2019;s benchmark index (<xref ref-type="bibr" rid="ref10">Jamak, 2018</xref>; <xref ref-type="bibr" rid="ref18">Seng and Thaker, 2018</xref>). These derivatives attract diverse participants who employ varied investment strategies, with 25% of publicly listed firms on Bursa Malaysia using derivatives for hedging from 2003 to 2007 (<xref ref-type="bibr" rid="ref18">Seng and Thaker, 2018</xref>).</p>
<p>In financial markets, customer segmentation divides populations into distinct groups based on common characteristics, enabling tailored services that help provide improved investment strategies (<xref ref-type="bibr" rid="ref5">Clark-Murphy and Soutar, 2005</xref>; <xref ref-type="bibr" rid="ref12">Kashwan and Velu, 2013</xref>; <xref ref-type="bibr" rid="ref26">Wood and Zaichkowsky, 2004</xref>). Trading segmentation identifies trader typologies based on strategies, risk tolerance, and behavioral patterns which include high-frequency traders, long-term investors, and hedgers (<xref ref-type="bibr" rid="ref13">Keller and Siegrist, 2006</xref>; <xref ref-type="bibr" rid="ref2">Bergl&#x00F6;f, 1985</xref>). While extensive research exists on investor segmentation in stock markets, limited studies have explored segmentation in derivatives markets. Given the differences in underlying assets, market behaviors, and risk dynamics, studying these segments is critical for understanding trading patterns in the derivatives market (<xref ref-type="bibr" rid="ref22">Subeesh and Liya, 2024</xref>; <xref ref-type="bibr" rid="ref21">Somanathan and Nageswaran, 2015</xref>). The resulting trader profiles can help regulators determine when trader positions become large enough to potentially manipulate prices away from legitimate supply and demand conditions (<xref ref-type="bibr" rid="ref17">Sanders et al., 2004</xref>).</p>
<p>This study applies K-means clustering to historical FCPO and FKLI trade data to identify distinct derivatives trader clusters, representing one of the first comprehensive applications of machine learning techniques specifically to derivatives trader segmentation. MacQueen in 1967 introduced the K-means algorithm as a method for partitioning observations into k clusters, establishing the mathematical framework that remains fundamental to modern clustering applications (<xref ref-type="bibr" rid="ref14">MacQueen, 1967</xref>). K-means efficiently partitions data into groups by minimizing intra-cluster variance, providing interpretable results for financial market analysis (<xref ref-type="bibr" rid="ref26">Wood and Zaichkowsky, 2004</xref>; <xref ref-type="bibr" rid="ref7">Fawaid Ridwan and Supian, 2021</xref>; <xref ref-type="bibr" rid="ref11">Kalra Sahi and Arora, 2012</xref>). The categorical variables within clusters were analyzed, and performance differences between clusters were investigated to provide insights into the behavior of derivatives market traders. The analysis also considered potential implications for trading strategies and risk management practices that could benefit market stakeholders. A novel decision tree validation approach is developed to uniquely characterize cluster membership, providing actionable behavioral insights for market stakeholders. This approach provides better identification compared to the current usage of ANOVA and Hypothesis testing that is used to highlight the differences between the resulted clusters, which does not require any normality assumptions and/or linearity.</p>
<p>The remainder of this paper is organized as follows: Section 2 presents a literature review of derivatives trading and clustering methodologies. Section 3 details data collection, feature engineering, and transformation techniques. Section 4 includes a detailed analysis of the K-means clustering outcomes across continuous and categorical variables. Section 5 explores cluster characteristics and suggests the main criteria of the trader clusters using decision tree node splits, validated through boxplot distributions. Section 6 summarizes the findings and offers future direction.</p>
</sec>
<sec id="sec2">
<title>Literature review</title>
<p>Research on derivatives investor behavior has revealed distinct trading preferences and patterns. <xref ref-type="bibr" rid="ref27">Yuen (2013)</xref> suggested that investment experience directly correlates with average returns and trading performance, while also identifying heterogeneous investor profiles characterized by different risk tolerances, holding periods, and product preferences. The shift toward online derivatives trading has also influenced investor behavior, with studies showing increased trading frequency and altered decision-making patterns among participants using digital platforms (<xref ref-type="bibr" rid="ref27">Yuen, 2013</xref>). <xref ref-type="bibr" rid="ref19">Shi et al. (2018)</xref> claimed that certain traders participate in specific transaction patterns, and only some trading characteristics of certain traders in a time window will reflect the trading behavior patterns. This suggested distinct behavioral clusters within derivatives markets.</p>
<p>Related studies that used trade data for clustering investors into significant groups were reviewed to demonstrate the application of clustering methodologies in financial market traders. Notably, there remains a significant shortage of research applying clustering techniques to derivatives market data. This study addresses this research gap by utilizing trade data extracted specifically from derivatives markets to identify distinct investor profiles, thereby extending the application of clustering methodologies beyond the commonly studied equity markets. Additionally, an innovative decision tree validation methodology for post-clustering validation and characterization provides decision rules for each identified cluster, representing a novel contribution to financial market segmentation research. The related work to this research is summarized in <xref ref-type="table" rid="tab1">Table 1</xref>.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Summary of related work.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Authors</th>
<th align="center" valign="top">Year</th>
<th align="left" valign="top">Dataset</th>
<th align="left" valign="top">Algorithm</th>
<th align="left" valign="top">Clusters</th>
<th align="left" valign="top">Ref.</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Shin &#x0026; Sohn</td>
<td align="center" valign="top">2004</td>
<td align="left" valign="top">2,999 customers (HTS &#x0026; assisted trading)</td>
<td align="left" valign="top">K-means, SOM, Fuzzy K-means</td>
<td align="left" valign="top">Normal (95%), Best (3%), VIP (0.2&#x2013;0.5%)</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref20">Shin and Sohn (2004)</xref>
</td>
</tr>
<tr>
<td align="left" valign="top">Wang et al.</td>
<td align="center" valign="top">2009</td>
<td align="left" valign="top">30,287 investors</td>
<td align="left" valign="top">Voting K-means</td>
<td align="left" valign="top">Conservative (52%), Speculative (27%), Moderate (21%)</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref25">Wang et al. (2009)</xref>
</td>
</tr>
<tr>
<td align="left" valign="top">Goshima et al.</td>
<td align="center" valign="top">2019</td>
<td align="left" valign="top">144 trading desks</td>
<td align="left" valign="top">Hierarchical Clustering</td>
<td align="left" valign="top">HFT Market Makers (8%), Opportunistic HFT (17%), Middle-Frequency (74%)&#x202F;+&#x202F;Low-Frequency (manual)</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref8">Goshima et al. (2019)</xref>
</td>
</tr>
<tr>
<td align="left" valign="top">Thompson et al.</td>
<td align="center" valign="top">2021</td>
<td align="left" valign="top">52,025 accounts</td>
<td align="left" valign="top">K-prototype</td>
<td align="left" valign="top">Active (19%), Early Savers (36%), Just-In-Time (27%), Older (7%), Systematic Savers (12%)</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref23">Thompson et al. (2021)</xref>
</td>
</tr>
<tr>
<td align="left" valign="top">Hwang et al.</td>
<td align="center" valign="top">2024</td>
<td align="left" valign="top">339,007 investors, 955,035 entries</td>
<td align="left" valign="top">Gaussian Mixture Model</td>
<td align="left" valign="top">8 clusters (unspecified)</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref9">Hwang et al. (2024)</xref>
</td>
</tr>
<tr>
<td align="left" valign="top">Vlahavas et al.</td>
<td align="center" valign="top">2024</td>
<td align="left" valign="top">105,589,345 transactions</td>
<td align="left" valign="top">K-means</td>
<td align="left" valign="top">Cluster 1 (61.4%), Cluster 2 (19.3%), Cluster 3 (11.5%), Cluster 4 &#x0026; 5 (~3% each)</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref24">Vlahavas et al. (2024)</xref>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Shin and Sohn focused on total trade amounts over three months, analyzing representative-assisted trading and the online Home Trading System (HTS) of 2,999 customers (<xref ref-type="bibr" rid="ref20">Shin and Sohn, 2004</xref>). The authors applied K-means, Self Organizing Maps, and fuzzy K-means as the clustering algorithms. The representative-assisted trading data were described by &#x201C;total trade amount&#x201D; and &#x201C;representative-assisted trade amount.&#x201D; The online HTS was represented by &#x201C;total trade amount&#x201D; and &#x201C;trade amount in HTS&#x201D; as the main raw features. The authors identified three clusters, normal customers (95%) (trading below specified thresholds in both trading modes), best customers (3%) (trading at intermediate levels), and VIP Customers (0.2&#x2013;0.5%) who exhibited the highest trade volumes across both trading modes. The authors introduced a new brokerage commission policy based on the identified clusters for a potential of higher profit.</p>
<p>Wang et al. used the records of 30,287 investors to categorize them into three predefined clusters. The customer purchasing and selling frequency, proportion of transaction amount to total assets, and proportion of deposit to total assets were used as the clustering features (<xref ref-type="bibr" rid="ref25">Wang et al., 2009</xref>). The authors used voting K-means to categorize the investors into three groups, Conservative Investors (52%), Speculative Investors (27%), and Moderate Investors (21%). Conservative Investors preferred low-risk instruments like time deposits, demonstrating minimal engagement with high-risk products. Speculative Investors favored high-risk financial products across all categories. While Moderate Investors adopted a balanced approach, blending conservative and speculative strategies. The clusters acquired an accuracy of 87% when evaluated using a randomly selected 200 customers from the dataset.</p>
<p>Goshima et al. analyzed 144 trading desks using hierarchical clustering based on four key metrics, Cancellation to Order Ratio, Inventory Ratio, Number of Actions per Stock, and Number of Stocks per Trading Desk (<xref ref-type="bibr" rid="ref8">Goshima et al., 2019</xref>). Their analysis initially yielded ten clusters, which they subsequently consolidated into three main trader categories with distinctive characteristics. The High-Frequency Trader Market Makers (8%) exhibited the highest cancellation-to-order ratios combined with minimal inventory holdings. Investors in this group are typical for high-frequency limit order strategies. The Opportunistic High-Frequency Traders (17%) displayed either elevated cancellation-to-order ratios or reduced inventory ratios, but not both simultaneously. The Middle-Frequency Traders (74%) maintained moderate values across both the Cancellation-to-Order Ratio and the Inventory Ratio. Additionally, they included a fourth category outside their clustering analysis (Low-Frequency Traders) which comprised of additional 2,177 trading desks with distinctly different trading patterns.</p>
<p>Thompson et al. used K-prototype clustering to segment 52,025 accounts based on investor demographics, trading frequency, and traded amount. Their analysis resulted in five clusters: Active Traders (19%), engaging in frequent, high-volume manual trades with moderate risk tolerance; Early Savers (36%), younger individuals relying on systematic transactions with minimal trading activity; Just-In-Time (27%), characterized by infrequent, small manual trades with slightly lower risk tolerance; Older Investors (7%), who prioritized withdrawals and dividends and exhibited the lowest risk tolerance; and Systematic Savers (12%), who employed periodic, systematic trading with a similar risk profile to active traders (<xref ref-type="bibr" rid="ref23">Thompson et al., 2021</xref>).</p>
<p>Hwang et al. conducted an investor clustering analysis using a substantial dataset comprising 339,007 investors with 955,035 data entries spanning January 2016 to December 2020. They utilized 23 variables across five categories: account overview, buy/sell orders, deposits/withdrawals, transaction proportions, and transaction details. The researchers identified eight clusters by employing the Gaussian Mixture Model. The resulting clusters exhibited varying characteristics, including differences in average balances, trading volumes, transaction values, turnover rates, and deposit/withdrawal patterns (<xref ref-type="bibr" rid="ref9">Hwang et al., 2024</xref>). Notably, some clusters demonstrated inverse relationships between account balance and trading activity, while others showed distinctive patterns in terms of transaction frequency and value.</p>
<p>Vlahavas et al. analyzed Bitcoin transaction behavior using K-means clustering on a comprehensive dataset comprising 105,589,345 transactions to identify distinct user behavioral patterns in cryptocurrency markets (<xref ref-type="bibr" rid="ref24">Vlahavas et al., 2024</xref>). Their analysis resulted in five clusters: Cluster 1 (61.4%), Cluster 2 (19.3%), Cluster 3 (11.5%), and Clusters 4 and 5 (approximately 3% each). While the study did not provide detailed names for each cluster, it demonstrated the effectiveness of unsupervised clustering techniques in revealing hidden patterns within blockchain transaction data, providing insights into the heterogeneous nature of cryptocurrency market participants (<xref ref-type="bibr" rid="ref24">Vlahavas et al., 2024</xref>).</p>
</sec>
<sec sec-type="methods" id="sec3">
<title>Methodology</title>
<p>The stepwise framework of the proposed methodology is illustrated in <xref ref-type="fig" rid="fig1">Figure 1</xref>.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Methodology flow chart.</p>
</caption>
<graphic xlink:href="frai-08-1640776-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart detailing a data analysis process. Steps include Data Collection, Data Pre-processing, Feature Engineering (with factors like Total Number of Trades and Average ROI), IHS Data Transformation, and K-means Clustering. It connects to Clustering Performance Metrics with Elbow Method and Silhouette Score. The process continues with Determination of Optimal Number of Clusters, Cluster Identification (using Decision Tree Splitting Conditions), Cluster Characterization, and concludes with an Intervention Plan.</alt-text>
</graphic>
</fig>
<sec id="sec4">
<title>Data description</title>
<p>The data used for this analysis were provided by BURSA Malaysia. The dataset comprises 11,222,606 rows of trade data collected between January 2022 and December 2022, covering two product codes, FCPO and FKLI. The records are stored in a structured SQL database. The data was filtered to include only traders registered in Malaysia, which reduced it to 11,117,203 rows, removing approximately 1% of the original data. To facilitate efficient data management and querying, a unique primary key was created by hashing a combination of three attributes: investor ID, broker participant ID, and account ID. The hierarchical structure prioritizes the broker participant ID, followed by the investor ID, and finally the account ID. This process generated 9,852 unique primary hash keys, representing 9,852 unique accounts, 8,816 unique traders, and 13 broker participants.</p>
<p>The dataset was further categorized based on the frequency of trade records for FCPO and FKLI to capture the trading preferences. Each unique primary hash key was classified into one of five categories: FCPO dominant, FCPO favored, neutral, FKLI favored, and FKLI dominant. Each transaction is made using one of two different trading strategy types, SPD (Derivatives that are based on the spread between the prices of two or more assets) and NRM (derivatives with one directional to buy/sell contracts). Therefore, the most frequent strategy type used by each unique primary hash key was recorded and associated with the respective account. Five categorical variables describe investor traits in the dataset, age group, gender, investor type, trade product preference, and most used strategy type. The characteristics of the studied data are illustrated in <xref ref-type="fig" rid="fig2">Figure 2</xref>.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Distribution of categorical variables across the dataset of 9,773 derivatives traders (79 traders were removed due to no realized profit).</p>
</caption>
<graphic xlink:href="frai-08-1640776-g002.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Bar charts depicting investor demographics and trading preferences. The Investor Type chart shows Retail at 9,675, Local at 78, and RTIP at 20. The Gender chart has Male at 7,849 and Female at 1,924. The Age Band chart shows most investors in the 36&#x2013;45 range at 3,225, decreasing with age. The Trading Preference chart highlights FCPO dominant at 7,317 and FKLI dominant at 1,931. The Main Strategy Type chart shows NRM at 8,640 and SPD at 1,133.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec5">
<title>Feature engineering and transformation</title>
<p>Six features were generated to analyze the trading behavior of traders. Each feature was designed to capture a distinct aspect of the trading activities. The generated features are the Total Number of Trades, Total Traded Amount, Overall Realized Profit, Average Return on Investment (ROI), Maximum Account Vintage, and Median Holding Days. Each feature is derived from the dataset and grouped by the unique primary hash key index, ensuring that they accurately represent individual trading behavior.</p>
<p>The Total Number of Trades corresponds to the total count of trade records in the dataset associated with each unique primary hash key. The Total Traded Amount is calculated as the cumulative sum of the trade values for all buy and sell transactions grouped by the unique primary hash key. The Overall Realized Profit represents the net profit or loss achieved by each trader. It is calculated by subtracting the total bought amount from the total sold amount for matched trades, where the quantities of bought and sold transactions align. A positive value indicates a net profit, while a negative value signifies a loss. Similarly, Average ROI is calculated as the realized profit divided by the total bought amount for matched trades. This metric provides a normalized measure of profitability, allowing for direct comparisons across traders regardless of the scale of their trading activities. It is particularly useful for identifying efficient traders who achieve high returns with limited resources. The Maximum Account Vintage reflects the longevity of a trader&#x2019;s account and is calculated as the difference between the last recorded trade date and the account creation date (expressed in years). The account vintage provides insights into the trader&#x2019;s experience and commitment over time to distinguish between newer participants and long-standing traders who may exhibit more stable or sophisticated trading behaviors. Finally, Median Holding Days capture the typical duration for which a trader holds a trade position before closing it. This is calculated as the median of the holding durations for all matched trades associated with each unique primary hash key, where the holding duration is the time elapsed between the creation of the buy and sell transactions. It offers a view into the trader&#x2019;s trading strategy, revealing whether they tend toward short-term trading for quick gains or long-term investments that aim for sustained returns. The formulated engineered features are summarised in <xref ref-type="table" rid="tab2">Table 2</xref>.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Formulated engineered features.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Feature name</th>
<th align="left" valign="top">Description</th>
<th align="left" valign="top">Formula</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Total number of trades</td>
<td align="left" valign="middle">Total count of trades per trader (per hash key)</td>
<td align="left" valign="middle">
<inline-formula>
<mml:math id="M1">
<mml:msub>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:msubsup>
<mml:mn>1</mml:mn>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left" valign="middle">Total traded amount</td>
<td align="left" valign="middle">The sum of all trade values (buy and sell) for each trader</td>
<td align="left" valign="middle">
<inline-formula>
<mml:math id="M2">
<mml:msub>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:msubsup>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="italic">ij</mml:mi>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi mathvariant="italic">ij</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left" valign="middle">Overall realized profit</td>
<td align="left" valign="middle">Net profit/loss from matched trades</td>
<td align="left" valign="middle">
<inline-formula>
<mml:math id="M3">
<mml:msub>
<mml:mi>ORP</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mo movablelimits="false">&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:munderover>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi mathvariant="italic">ik</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi mathvariant="italic">ik</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left" valign="middle">Average ROI</td>
<td align="left" valign="middle">Normalized profitability measure per trader</td>
<td align="left" valign="middle">
<inline-formula>
<mml:math id="M4">
<mml:msub>
<mml:mi mathvariant="italic">ROI</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>ORP</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:munderover>
<mml:mo movablelimits="false">&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:munderover>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi mathvariant="italic">ik</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left" valign="middle">Maximum account vintage</td>
<td align="left" valign="middle">Trader&#x2019;s account age in years</td>
<td align="left" valign="middle">
<inline-formula>
<mml:math id="M5">
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">LTD</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">ACD</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mn>365</mml:mn>
</mml:mfrac>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left" valign="middle">Median holding days</td>
<td align="left" valign="middle">Median number of days trades are held before selling.</td>
<td align="left" valign="middle">
<inline-formula>
<mml:math id="M6">
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext mathvariant="italic">Median</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi mathvariant="italic">ik</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>B</mml:mi>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi mathvariant="italic">ik</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</inline-formula>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><inline-formula>
<mml:math id="M7">
<mml:mi mathvariant="normal">i</mml:mi>
</mml:math>
</inline-formula>: index of trader, <inline-formula>
<mml:math id="M8">
<mml:mi mathvariant="normal">j</mml:mi>
</mml:math>
</inline-formula>: index of trade, <inline-formula>
<mml:math id="M9">
<mml:mi mathvariant="normal">k</mml:mi>
</mml:math>
</inline-formula>: index of matched trade, <inline-formula>
<mml:math id="M10">
<mml:msub>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>: number of rows for trader <inline-formula>
<mml:math id="M11">
<mml:mi mathvariant="normal">i</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M12">
<mml:msub>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>: total number of match traders for trader <inline-formula>
<mml:math id="M13">
<mml:mi mathvariant="normal">i</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M14">
<mml:msub>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi>ij</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>: the price of trade <inline-formula>
<mml:math id="M15">
<mml:mi mathvariant="normal">j</mml:mi>
</mml:math>
</inline-formula> for trader <inline-formula>
<mml:math id="M16">
<mml:mi mathvariant="normal">i</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M17">
<mml:msub>
<mml:mi mathvariant="normal">Q</mml:mi>
<mml:mi>ij</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>: quantity of trade <inline-formula>
<mml:math id="M18">
<mml:mi mathvariant="normal">j</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M19">
<mml:msub>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi>ik</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>: total value of sell trade, <inline-formula>
<mml:math id="M20">
<mml:msub>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>ik</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>: total value of buy trade, <inline-formula>
<mml:math id="M21">
<mml:msub>
<mml:mi>LTD</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>: last trade date for trader <inline-formula>
<mml:math id="M22">
<mml:mi mathvariant="normal">i</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M23">
<mml:msub>
<mml:mi>ACD</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>: account creation date for trader <inline-formula>
<mml:math id="M24">
<mml:mi mathvariant="normal">i</mml:mi>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math id="M25">
<mml:msub>
<mml:mi>SD</mml:mi>
<mml:mi>ik</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>: sell date of matched trade <inline-formula>
<mml:math id="M26">
<mml:mi mathvariant="normal">k</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M27">
<mml:msub>
<mml:mi>BD</mml:mi>
<mml:mi>ik</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>: buy date of matched trade <inline-formula>
<mml:math id="M28">
<mml:mi mathvariant="normal">k</mml:mi>
</mml:math>
</inline-formula>.</p>
</table-wrap-foot>
</table-wrap>
<p>A total of 79 unique primary hash keys had no realized profit being computed as no match trades were found. As a result, those hash keys were removed from the analysis which reduced the data from 9,852 to 9,773 unique rows (representing only 0.8% of the dataset). The deleted entries contribute to only 334 total trades in the data (approximately 0.003% of the entire dataset), where 32 out of these 79 unique primary hash keys have exactly one trade from the entire Jan 2022 to Dec 2022 period.</p>
<p><xref ref-type="fig" rid="fig3">Figure 3</xref> illustrates the preprocessing flow chart of the variables and engineered features.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Data preprocessing flow chart of variables and features.</p>
</caption>
<graphic xlink:href="frai-08-1640776-g003.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart depicting a data processing framework. The process begins by creating a primary hash key using Broker Participant ID, Investor ID, Account ID, and Malaysian, generating 9,852 rows. Data is grouped by each primary hash key. Categorical variables include trading preferences (FCPO Dominant, FCPO Favored, Neutral, FKLI Favored, FKLI Dominant) and main strategy type (SPD, NRM). Feature engineering involves calculating total number of trades, traded amount, realized profit, average ROI, maximum account vintage, and median holding days, using formulas from Table 2. After removing 79 keys with no realized profit, the dataset reduces to 9,773 rows.</alt-text>
</graphic>
</fig>
<p>These features exhibit substantial variability due to the presence of outliers, which can distort the clustering performance of the K-means model. To address this issue, the Inverse Hyperbolic Sine (IHS) transformation was applied to scale down the values of features with extremely large ranges while preserving the overall distribution structure (<xref ref-type="disp-formula" rid="EQ3">Equation 1</xref>). First, IHS transformation accommodates the full range of financial data including zero and negative values without requiring data truncation or sign loss, unlike log transformation which necessitates positive-only inputs. Second, IHS provides asymptotic behavior that compresses extreme outliers through the asymptotic linearity property, effectively reducing outlier leverage while retaining their directional information (<xref ref-type="disp-formula" rid="EQ4">Equation 2</xref>). Finally, unlike other normalization techniques such as log transformation or min-max scaling, Hence, IHS maintains the distributional properties and relative ordering of observations while mitigating the disproportionate influence of outliers (<xref ref-type="bibr" rid="ref1">Bellemare and Wichman, 2020</xref>; <xref ref-type="bibr" rid="ref3">Burbidge et al., 1988</xref>).<disp-formula id="EQ3">
<label>(1)</label>
<mml:math id="M29">
<mml:mo>sin</mml:mo>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mo>ln</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>+</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msqrt>
<mml:mo stretchy="true">)</mml:mo>
<mml:mspace width="0.25em"/>
</mml:math>
</disp-formula><disp-formula id="EQ4">
<label>(2)</label>
<mml:math id="M30">
<mml:munder>
<mml:mo>lim</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mo>+</mml:mo>
<mml:mo>&#x221E;</mml:mo>
</mml:mrow>
</mml:munder>
<mml:mfrac>
<mml:mrow>
<mml:mo>sin</mml:mo>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>ln</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mspace width="0.25em"/>
</mml:math>
</disp-formula></p>
</sec>
<sec id="sec6">
<title>K-means clustering</title>
<p>The six generated features were used in the K-means clustering algorithm. K-means is a simple yet powerful clustering algorithm that can be used over a <inline-formula>
<mml:math id="M31">
<mml:mi>d</mml:mi>
</mml:math>
</inline-formula>-dimensional vector <inline-formula>
<mml:math id="M32">
<mml:mi>X</mml:mi>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">[</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo stretchy="true">]</mml:mo>
</mml:math>
</inline-formula> in <inline-formula>
<mml:math id="M33">
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>d</mml:mi>
</mml:msup>
</mml:math>
</inline-formula>. For a set of <inline-formula>
<mml:math id="M34">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> input data points the k-means algorithm begins with the initialization of <inline-formula>
<mml:math id="M35">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula> centroids, <inline-formula>
<mml:math id="M36">
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>, which are randomly selected from the set of data <inline-formula>
<mml:math id="M37">
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula>. These centroids represent the initial cluster centers (step 1).</p>
<p>Each data point <inline-formula>
<mml:math id="M38">
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> in <inline-formula>
<mml:math id="M39">
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula> is then assigned to the nearest centroid <inline-formula>
<mml:math id="M40">
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> based on a distance metric (typically the Euclidean distance given in <xref ref-type="disp-formula" rid="EQ1">Equation 3</xref>) as the assignment step (step 2).<disp-formula id="EQ1">
<label>(3)</label>
<mml:math id="M41">
<mml:mi>d</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>d</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi mathvariant="italic">il</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi mathvariant="italic">jl</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:math>
</disp-formula></p>
<p>Each data point <inline-formula>
<mml:math id="M42">
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is assigned to the cluster <inline-formula>
<mml:math id="M43">
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> whose centroid <inline-formula>
<mml:math id="M44">
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is the closest. Once all data points have been assigned to clusters, the centroids <inline-formula>
<mml:math id="M45">
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> are recalculated as the mean of the data points in each cluster as given in <xref ref-type="disp-formula" rid="EQ2">Equation 4</xref>:<disp-formula id="EQ2">
<label>(4)</label>
<mml:math id="M46">
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mspace width="0.25em"/>
</mml:math>
</disp-formula></p>
<p>Where <inline-formula>
<mml:math id="M47">
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
</mml:math>
</inline-formula> is the number of data points in cluster <inline-formula>
<mml:math id="M48">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula>, and the sum is over all data points assigned to cluster <inline-formula>
<mml:math id="M49">
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>. Hence, all the centroids are now updated (step 3). The steps of assignment (step 2) and update (step 3) are repeated until the centroids are no longer changing significantly, or until a maximum number of iterations is reached. The convergence criterion can be the change in the centroids&#x2019; positions or the change in the cluster assignments. The objective of k-means is to minimize the within-cluster sum of squares (WCSS), also known as the inertia, as shown in <xref ref-type="disp-formula" rid="EQ5">Equation 5</xref>:<disp-formula id="EQ5">
<label>(5)</label>
<mml:math id="M50">
<mml:mi>J</mml:mi>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:math>
</disp-formula></p>
<p>The objective function <inline-formula>
<mml:math id="M51">
<mml:mi>J</mml:mi>
</mml:math>
</inline-formula> quantifies the total variance within the clusters, and the k-means algorithm seeks to minimize this value. The final output is a set of <inline-formula>
<mml:math id="M52">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula> clusters <inline-formula>
<mml:math id="M53">
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</inline-formula> and their corresponding centroids <inline-formula>
<mml:math id="M54">
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="ref14">MacQueen, 1967</xref>).</p>
</sec>
</sec>
<sec sec-type="results" id="sec7">
<title>Results and discussion</title>
<p>The optimal number of clusters (k) was determined using four methods, the Elbow Method, Silhouette Score, Davies-Bouldin Index, and Calinski-Harabasz Index (<xref ref-type="bibr" rid="ref16">Rousseeuw, 1987</xref>; <xref ref-type="bibr" rid="ref6">Davies and Bouldin, 2009</xref>; <xref ref-type="bibr" rid="ref4">Cali&#x0144;ski and Harabasz, 1974</xref>). These methods collectively identified the optimal range of k from 3 to 6, where the Elbow Method shows a noticeable bend at k&#x202F;=&#x202F;5, the Silhouette Score reaches a local maximum, the Davies-Bouldin Index achieves a local minimum value, and the Calinski-Harabasz Index demonstrates high values at this point as illustrated in <xref ref-type="fig" rid="fig4">Figure 4</xref>. The convergence of these four indicators at k&#x202F;=&#x202F;5 provides strong statistical evidence for this optimal cluster number. The five identified clusters are well-separated in the reduced two-dimensional PCA space as illustrated in <xref ref-type="fig" rid="fig5">Figure 5</xref>. The distribution of data points among the five identified clusters is given in <xref ref-type="fig" rid="fig6">Figure 6</xref>.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Cluster validation metrics for determining optimal number of clusters.</p>
</caption>
<graphic xlink:href="frai-08-1640776-g004.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Four line graphs depict different methods for determining the optimal number of clusters. The Elbow Method graph shows a sharp decrease in WSS at 5 clusters. The Silhouette Score graph peaks at 5 clusters. The Davies-Bouldin Index graph reaches a low at 5 clusters. The Calinski-Harabasz Index graph peaks at 5 clusters. Vertical dashed lines mark potential optimal clusters, highlighting 5 as a common point.</alt-text>
</graphic>
</fig>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>K-means clustering results visualized in two-dimensional PCA space.</p>
</caption>
<graphic xlink:href="frai-08-1640776-g005.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Scatter plot showing K-Means clustering visualization with data points grouped into five clusters, each represented by a different color: blue, orange, green, red, and purple. The axes are labeled PCA1 and PCA2.</alt-text>
</graphic>
</fig>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Cluster size distribution showing trader count and percentage for each identified cluster.</p>
</caption>
<graphic xlink:href="frai-08-1640776-g006.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Histogram showing cluster label frequencies. Label 1 has 1992 (20.4%), label 2 has 1427 (14.6%), label 3 has 2065 (21.1%), label 4 has 1169 (12.0%), and label 5 has 3120 (31.9%).</alt-text>
</graphic>
</fig>
<p><xref ref-type="fig" rid="fig7">Figure 7</xref> shows the distributions of the six features across the five clusters. The Total Number of Trades and Total Traded Amount exhibit wide ranges across clusters, with clusters 1 and 3 showing higher medians compared to clusters 2 and 4. Similarly, Maximum Account Vintage Years in clusters 2 and 4 demonstrate wider interquartile ranges compared to others. The Total Realized Profit and Average ROI in clusters 1 and 4 display more negative values, while cluster 3 exhibits higher positive medians. The Median Holding Days feature remains relatively low across all clusters, with slightly longer durations observed in clusters 2 and 4.</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Distribution of six engineered features across the five trader clusters (outliers removed for clarity).</p>
</caption>
<graphic xlink:href="frai-08-1640776-g007.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Box plots showing pre-transformation feature distributions by cluster without outliers. Plots include total number of trades, total traded amount, maximum account vintage years, total realized profit, average ROI, and median number of holding days. Each plot compares five clusters labeled 1 to 5, distinguished by color.</alt-text>
</graphic>
</fig>
<p>The distribution of the categorical variables across clusters and the distinct patterns in the trader profiles are found in <xref ref-type="fig" rid="fig8">Figures 8</xref>, <xref ref-type="fig" rid="fig9">9</xref>. Retail remains the main investor type across all clusters, while RTIP and Local investors are mostly identified in Cluster 1 and 3. The proportion of males to females traders amongst all clusters is estimated roughly as 80 to 20 percent, with cluster 3 having the highest female percentage at 22%. The distribution of the Main Strategy Type exhibits a similar trend as well. Around 83&#x2013;90% of the traders are mainly using the NRM strategy, with cluster 3 having a slightly higher percentage of using SPD as their main strategy.</p>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>Categorical variable distributions across trader clusters showing investor type and gender.</p>
</caption>
<graphic xlink:href="frai-08-1640776-g008.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Four charts detail the distribution of investors and gender across clusters. The top left chart shows counts of investor types across five clusters, mostly retail in Cluster 5. The top right chart illustrates cluster distribution within investor types, with Cluster 1 dominating retail. The bottom left chart shows gender distribution across clusters, primarily male in Cluster 5. The bottom right chart highlights cluster distribution within gender, with Cluster 1 prevalent among both males and females.</alt-text>
</graphic>
</fig>
<fig position="float" id="fig9">
<label>Figure 9</label>
<caption>
<p>Categorical variable distributions across trader clusters showing strategy type, age band, and trading preference.</p>
</caption>
<graphic xlink:href="frai-08-1640776-g009.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">The image contains six bar charts illustrating distributions of various factors across clusters. The first chart shows the distribution of main strategy types (NRM and SPD) across different clusters. The second chart displays cluster distribution within these strategy types. The third chart represents the distribution of age bands across clusters. The fourth chart details cluster distribution within each age band. The fifth chart shows distribution of trading preferences (e.g., FKLI and FCPO) across clusters, while the sixth chart illustrates cluster distribution within each trading preference category. Each chart includes detailed percentages and count data.</alt-text>
</graphic>
</fig>
<p>As of the age bands, Cluster 5, comprises a significant proportion of younger traders (18&#x2013;25&#x202F;years and 26&#x2013;35&#x202F;years). This is also shown by the dominant lower Maximum Account Vintage value compared to other clusters. Cluster 2 shows an opposite age band trend compared to Cluster 5, where experienced and elder traders are concentrated there. Finally, the age groups in Cluster 1 are uniformly distributed. The trading preference within clusters shows that traders in Clusters 1, 3, and 5 are more inclined to trade FCPO when compared to Cluster 2 and 4 traders who mostly favor FKLI as their main trading preference.</p>
<p>A decision tree classifier is employed to determine the splitting mechanism that differentiates each cluster from the others. The pruned decision trees reveal the primary criteria for identifying and characterizing each cluster based on their trading behaviors and features (<xref ref-type="supplementary-material" rid="SM1">Appendix A</xref>). The characteristics of each cluster are detailed as follows.</p>
<p>Cluster 1 (representing 20.4% of the dataset) is defined by high trading activity, substantial traded amounts, and short holding periods. However, traders in Cluster 1 achieved consistently negative profits, indicating a high-risk and high-turnover trading strategy. The primary conditions distinguishing this cluster are a Total Realized Profit of less than RM-51.85&#x202F;k, a Total Traded Amount above RM17.57 million, and a Median Number of Holding Days under 2.22&#x202F;days, emphasizing frequent trading without consistent profitability.</p>
<p>Cluster 2 (representing 14.6% of the dataset) represents a low-risk, cautious trader profile achieving modest returns. The cluster is characterized by an Average ROI greater than 0.139%, a Median Holding Days above 0.299&#x202F;days, and a Total Number of Trades below 364.5, highlighting a conservative trading strategy with moderate engagement and consistent positive returns.</p>
<p>Cluster 3 (representing 21.1% of the dataset) is a high-activity, high-gain trader cluster, exhibiting substantial profits and short holding periods. Traders in this group have achieved a Total Realized Profit exceeding RM40.938 thousand, a Total Traded Amount above RM15.882 million, and a Median Holding Days under 0.993&#x202F;days, suggesting quick, successful trades with significant market engagement.</p>
<p>Cluster 4 (representing 12% of the dataset) represents cautious traders with low returns, reflecting a risk-averse strategy. This cluster is distinguished by traders achieving an Average ROI below &#x2212;0.205%, a Median Holding Days above 0.203&#x202F;days, and a Total Number of Trades below 306.5, indicating a conservative approach with limited profitability.</p>
<p>Finally, Cluster 5 (representing 31.9% of the dataset) encompasses low-activity and low-risk traders (potentially less experienced in the market). The defining criteria include a Total Number of Trades below 156.5, a Median Holding Days under 0.068&#x202F;days, a Maximum Account Vintage under 3.337&#x202F;years, and a Total Realized Profit under RM40.487 thousand, signifying cautious or infrequent trading and limited market exposure.</p>
<p>As a conclusion, the decision tree classifier results align closely with the observed boxplot distributions of the features across clusters as visualized in <xref ref-type="fig" rid="fig6">Figure 6</xref>. The matching patterns provide confidence that the features selected by the classifier effectively describe the unique characteristics of each cluster, enhancing the overall reliability of the analysis. Subsequently, the splitting conditions and trading behaviors revealed by the decision tree classifiers were used to label the identified clusters, providing clear and meaningful descriptions for each group. These labels capture the essence of the trading strategies, trading volume and frequency, and overall returns exhibited by the traders within each group. The labels of each identified cluster are detailed as follows.</p>
<p>As a conclusion, the decision tree classifier results align closely with the observed boxplot distributions of the features across clusters as visualized in <xref ref-type="fig" rid="fig6">Figure 6</xref>. The matching patterns provide confidence that the features selected by the classifier effectively describe the unique characteristics of each cluster, enhancing the overall reliability of the analysis. Subsequently, the splitting conditions and trading behaviors revealed by the decision tree classifiers were used to label the identified clusters, providing clear and meaningful descriptions for each group. These labels capture the essence of the trading strategies, the trading volume and frequency, and the overall returns exhibited by the traders within each group. The labels of each identified cluster are detailed as follows.</p>
<p>Cluster 1, &#x201C;High-Frequency, High-Risk Derivative Traders with Consistent Losses,&#x201D; represents a group of high-activity, high-turnover traders who prioritize frequent, short-term trades in hopes of achieving rapid gains. However, these traders often experience consistent losses. This group embodies a risk-heavy approach, where high trading volumes fail to translate into positive financial outcomes. They are also inclined to trade in FCPO products. Regulators could implement targeted interventions such as mandatory cooling-off periods or enhanced margin requirements for this segment, while brokers might benefit from automated risk controls and educational interventions to address their systematic losses. It seems to be a unique profile not commonly identified in equity market studies.</p>
<p>Cluster 2, &#x201C;Conservative, Steady-Growth Derivative Traders,&#x201D; is characterized by conservative, low-risk traders who achieve steady, modest returns, which is similar to &#x201C;Conservative Investors&#x201D; identified by Wang et al. These traders demonstrate a focus on minimizing risk, as reflected by their higher Average ROI, the longer Median Holding Days with low trade amounts. This group represents disciplined traders who prefer calculated strategies, avoiding excessive risk while ensuring a consistent, positive financial performance. Their cautious engagement reflects a preference for stable growth over aggressive expansion. They are also inclined to trade FKLI products with older traders. This segment presents opportunities for brokers to develop long-term investment products and advisory services, given their disciplined approach and consistent positive performance.</p>
<p>Cluster 3, &#x201C;High-Frequency, High-Yield Derivative Traders,&#x201D; comprises highly active, short-term traders who excel in generating substantial profits through quick trades. With a higher Traded Amount and shorter Median Holding Days, these traders successfully capitalize on rapid market movements as shown by their higher Realized Profit than other clusters. This group represents dynamic, successful traders who navigate the market with agility and precision, and similar to the &#x201C;Active Traders&#x201D; cluster described by Thompson et al. They are also inclined to trade FCPO products, while some have higher flexibility towards FKLI. Most RTIP and Local investors belong to this cluster as well. These high-performing traders could be offered premium services, lower transaction costs, and advanced trading tools by brokers, representing the most profitable client segment.</p>
<p>Cluster 4, &#x201C;Conservative, Low-Yield Derivative Traders,&#x201D; includes cautious traders who trade conservatively but yield low returns, shares similarities with &#x201C;Moderate Investors&#x201D; by Wang et al. Despite their steady approach (as indicated by negative Average ROI), they fail to achieve significant profitability. Their lower Number of Trades and longer Median Holding Days further reflect their preference for controlled and limited market engagement. This cluster captures the behavior of risk-averse traders who prioritize stability over aggressive strategies but struggle to convert this approach into meaningful financial gains. They are also inclined to trade in FKLI products. The negative average ROI despite cautious approaches may indicate market access barriers or information asymmetries that warrant regulatory attention and broker-provided educational support.</p>
<p>Cluster 5, &#x201C;Cautious, Low-Activity Novice Derivative Traders,&#x201D; represents low-activity traders who exhibit cautious behaviors, likely due to limited market experience. This group engages infrequently (as shown by their lower Number of Trades and Median Holding Days) while also having relatively short account histories (Maximum Account Vintage Years under 3.337&#x202F;years). Their lower Realized Profit reflects modest or limited financial outcomes as well. This cluster likely consists of newer or less-engaged traders (age 18&#x2013;35) who are still exploring the market or adopting a conservative approach to trading. They are also inclined to trade FCPO products. The predominance of younger, inexperienced traders in this largest cluster suggests the need for enhanced investor protection measures and mandatory financial literacy programs before derivatives trading authorization. This cluster is also similar to the &#x201C;Early Savers&#x201D; category identified by Thompson et al.</p>
</sec>
<sec id="sec8">
<title>Conclusions and future direction</title>
<p>This study explored the trading behaviors of traders in Bursa Malaysia&#x2019;s derivatives markets, with a specific focus on FCPO and FKLI products. While investor segmentation has been widely studied in stock markets, this study represents a breakthrough as one of the first to apply clustering techniques to investor behavior in the derivatives market. Through the application of K-means clustering on approximately 11 million trade records, five distinct clusters were identified, &#x201C;High-Frequency, High-Risk Derivative Traders with Consistent Losses,&#x201D; &#x201C;Conservative, Steady-Growth Derivative Trader,&#x201D; &#x201C;High-Frequency, High-Yield Derivative Traders,&#x201D; &#x201C;Conservative, Low-Yield Derivative Traders,&#x201D; and &#x201C;Cautious, Low-Activity Novice Derivative Traders.&#x201D; The methodological approach incorporated feature engineering and IHS transformation to address extreme data variability and outliers, thereby enhancing the robustness of the clustering algorithm. The details of the clusters were discussed deeply based on the characteristics identified using a novel decision tree approach and a thorough descriptive analysis.</p>
<p>Future research could incorporate questionnaire-based data to establish correlations between demographic characteristics, psychological traits, and trading behaviors to relate these attributes against the identified trader clusters for more insights. However, behavioral data integration through carefully designed questionnaires would require addressing privacy and regulatory constraints inherent in financial market research with nearly 10,000 traders. Additionally, expanding the analysis to include temporal dimensions through quarterly or semi-annual segmentation would facilitate an understanding of performance trends over time. Temporal clustering analysis could be conducted with larger datasets spanning multiple years and different market cycles to ensure sufficient trader activity across all seasons while maintaining statistical validity. Future studies could also developing methodologies to accurately calculate unrealized profits, which would provide a more comprehensive view of trader performance, particularly for long-term position holders. Cross-market validation using data from other emerging derivatives markets would also enhance the generalizability of findings beyond the Malaysian context. Lastly, the current analysis does not establish whether demographic characteristics influence trading behavior clustering or merely correlate with it. Future research should incorporate formal statistical testing and expand the demographic dataset as mentioned earlier to include variables such as education level, income, trading experience, and professional background to better understand the causal relationships between trader characteristics and behavioral patterns.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec9">
<title>Data availability statement</title>
<p>The datasets presented in this article are not readily available because the data used in this study were obtained under a data-sharing agreement with BURSA Malaysia. Due to the sensitive and proprietary nature of the trading data, access is restricted and the dataset is not publicly available. Interested researchers may request access directly from BURSA Malaysia; however, approval is subject to their discretion, and data access may involve administrative procedures and associated charges. The authors do not have the authority to share the dataset. Requests to access the datasets should be directed to ST, <email>tansieowyeek@bursamalaysia.com</email>.</p>
</sec>
<sec sec-type="author-contributions" id="sec10">
<title>Author contributions</title>
<p>ET: Visualization, Methodology, Validation, Writing &#x2013; original draft, Data curation. YH: Methodology, Funding acquisition, Writing &#x2013; review &#x0026; editing, Project administration, Supervision. HD: Funding acquisition, Writing &#x2013; review &#x0026; editing, Conceptualization, Validation. MA: Writing &#x2013; review &#x0026; editing, Validation, Data curation, Methodology. AA: Writing &#x2013; review &#x0026; editing, Methodology, Data curation, Validation. ST: Funding acquisition, Methodology, Writing &#x2013; review &#x0026; editing, Conceptualization, Validation.</p>
</sec>
<sec sec-type="funding-information" id="sec11">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. The authors would like to express their sincere gratitude to Universiti Teknologi PETRONAS (UTP) for the financial support provided under the research grant University Internal Research Funding (URIF) [cost center: 015LB0-094].</p>
</sec>
<ack>
<p>We also extend our deepest appreciation to BURSA Malaysia for granting access to a highly sensitive and high-value dataset containing over 11 million trade records. We are grateful for the continuous technical guidance and collaboration extended by the BURSA team, which significantly enhanced the quality and relevance of this work.</p>
</ack>
<sec sec-type="COI-statement" id="sec12">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec13">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec14">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec15">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2025.1640776/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/frai.2025.1640776/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bellemare</surname><given-names>M. F.</given-names></name> <name><surname>Wichman</surname><given-names>C. J.</given-names></name></person-group> (<year>2020</year>). <article-title>Elasticities and the inverse hyperbolic sine transformation</article-title>. <source>Oxf. Bull. Econ. Stat.</source> <volume>82</volume>, <fpage>50</fpage>&#x2013;<lpage>61</lpage>. doi: <pub-id pub-id-type="doi">10.1111/obes.12325</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Bergl&#x00F6;f</surname><given-names>E.</given-names></name></person-group> (<year>1985</year>). <source>A Note on the Typology of Financial Systems</source></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Burbidge</surname><given-names>J. B.</given-names></name> <name><surname>Magee</surname><given-names>L.</given-names></name> <name><surname>Robb</surname><given-names>A. L.</given-names></name></person-group> (<year>1988</year>). <article-title>Alternative transformations to handle extreme values of the dependent variable</article-title>. <source>J. Am. Stat. Assoc.</source> <volume>83</volume>, <fpage>123</fpage>&#x2013;<lpage>127</lpage>. doi: <pub-id pub-id-type="doi">10.1080/01621459.1988.10478575</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cali&#x0144;ski</surname><given-names>T.</given-names></name> <name><surname>Harabasz</surname><given-names>J.</given-names></name></person-group> (<year>1974</year>). <article-title>A dendrite method for cluster analysis</article-title>. <source>Commun. Stat. Theory Methods</source> <volume>3</volume>, <fpage>1</fpage>&#x2013;<lpage>27</lpage>.</citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clark-Murphy</surname><given-names>M.</given-names></name> <name><surname>Soutar</surname><given-names>G.</given-names></name></person-group> (<year>2005</year>). <article-title>Individual investor preferences: a segmentation analysis</article-title>. <source>J. Behav. Finance</source> <volume>6</volume>, <fpage>6</fpage>&#x2013;<lpage>14</lpage>. doi: <pub-id pub-id-type="doi">10.1207/s15427579jpfm0601_2</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Davies</surname><given-names>D. L.</given-names></name> <name><surname>Bouldin</surname><given-names>D. W.</given-names></name></person-group> (<year>2009</year>). <article-title>A cluster separation measure</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>2</volume>, <fpage>224</fpage>&#x2013;<lpage>227</lpage>.</citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fawaid Ridwan</surname><given-names>A.</given-names></name> <name><surname>Supian</surname><given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>IDX30 stocks clustering with K-means algorithm based on expected return and value at risk</article-title>. <source>Int. J. Quant. Res. Model.</source> <volume>2</volume>, <fpage>201</fpage>&#x2013;<lpage>208</lpage>.</citation></ref>
<ref id="ref8"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Goshima</surname><given-names>K.</given-names></name> <name><surname>Tobe</surname><given-names>R.</given-names></name> <name><surname>Uno</surname><given-names>J.</given-names></name></person-group>. <source>Trader classification by cluster analysis: interaction between HFTs and other traders</source>. (<year>2019</year>).</citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hwang</surname><given-names>Y.</given-names></name> <name><surname>Park</surname><given-names>J.</given-names></name> <name><surname>Kim</surname><given-names>J. H.</given-names></name> <name><surname>Lee</surname><given-names>Y.</given-names></name> <name><surname>Fabozzi</surname><given-names>F. J.</given-names></name></person-group> (<year>2024</year>). <article-title>Heterogeneous trading behaviors of individual investors: a deep clustering approach</article-title>. <source>Financ. Res. Lett.</source> <volume>65</volume>:<fpage>105481</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.frl.2024.105481</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Jamak</surname><given-names>F.</given-names></name></person-group> (<year>2018</year>) <source>Futures Market of Crude Palm Oil (FCPO) and Kuala Lumpur Index (FKLI) as the Price Discovery in Malaysia</source></citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kalra Sahi</surname><given-names>S.</given-names></name> <name><surname>Arora</surname><given-names>A. P.</given-names></name></person-group> (<year>2012</year>). <article-title>Individual investor biases: a segmentation analysis</article-title>. <source>Qual. Res. Financ. Mark.</source> <volume>4</volume>, <fpage>6</fpage>&#x2013;<lpage>25</lpage>. doi: <pub-id pub-id-type="doi">10.1108/17554171211213522</pub-id></citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kashwan</surname><given-names>K. R.</given-names></name> <name><surname>Velu</surname><given-names>C. M.</given-names></name></person-group> (<year>2013</year>). <article-title>Customer segmentation using clustering and data mining techniques</article-title>. <source>Int. J. Comput. Theor. Eng.</source>, <fpage>856</fpage>&#x2013;<lpage>861</lpage>. doi: <pub-id pub-id-type="doi">10.7763/IJCTE.2013.V5.811</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Keller</surname><given-names>C.</given-names></name> <name><surname>Siegrist</surname><given-names>M.</given-names></name></person-group> (<year>2006</year>). <article-title>Money attitude typology and stock investment</article-title>. <source>J. Behav. Finance</source> <volume>7</volume>, <fpage>88</fpage>&#x2013;<lpage>96</lpage>. doi: <pub-id pub-id-type="doi">10.1207/s15427579jpfm0702_3</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="other"><person-group person-group-type="author"><name><surname>MacQueen</surname><given-names>J.</given-names></name></person-group>, (<year>1967</year>). Multivariate observations. in Proceedings of the 5th Berkeley symposium on mathematical statistics and probability, 1, 281&#x2013;297.</citation></ref>
<ref id="ref15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rizal</surname><given-names>N.</given-names></name> <name><surname>Saifuddin</surname><given-names>S. A.</given-names></name> <name><surname>Abd Rahim</surname><given-names>S. H. A.</given-names></name> <name><surname>Mohd Nazri</surname><given-names>N.</given-names></name> <name><surname>Ab Aziz</surname><given-names>M. S.</given-names></name> <name><surname>Zainoddin</surname><given-names>A. I.</given-names></name></person-group> (<year>2023</year>). <article-title>The macroeconomic factors on Malaysia&#x2019;s future crude palm oil (FCPO)</article-title>. <source>Int. J. Acad. Res. Bus. Soc. Sci.</source> <volume>13</volume>. doi: <pub-id pub-id-type="doi">10.6007/ijarbss/v13-i3/16474</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rousseeuw</surname><given-names>P. J.</given-names></name></person-group> (<year>1987</year>). <article-title>Silhouettes: a graphical aid to the interpretation and validation of cluster analysis</article-title>. <source>J. Comput. Appl. Math.</source> <volume>20</volume>, <fpage>53</fpage>&#x2013;<lpage>65</lpage>. doi: <pub-id pub-id-type="doi">10.1016/0377-0427(87)90125-7</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sanders</surname><given-names>D. R.</given-names></name> <name><surname>Boris</surname><given-names>K.</given-names></name> <name><surname>Manfredo</surname><given-names>M.</given-names></name></person-group> (<year>2004</year>). <article-title>Hedgers, funds, and small speculators in the energy futures markets: an analysis of the CFTC'S commitments of traders reports</article-title>. <source>Energy Econ.</source> <volume>26</volume>, <fpage>425</fpage>&#x2013;<lpage>445</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.eneco.2004.04.010</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seng</surname><given-names>C. K.</given-names></name> <name><surname>Thaker</surname><given-names>H. M. T.</given-names></name></person-group> (<year>2018</year>). <article-title>Determinants of corporate hedging practices: Malaysian evidence</article-title>. <source>Rep. Econ. Finance.</source> <volume>4</volume>, <fpage>199</fpage>&#x2013;<lpage>220</lpage>. doi: <pub-id pub-id-type="doi">10.12988/ref.2018.8418</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname><given-names>G.</given-names></name> <name><surname>Ren</surname><given-names>L.</given-names></name> <name><surname>Miao</surname><given-names>Z.</given-names></name> <name><surname>Gao</surname><given-names>J.</given-names></name> <name><surname>Che</surname><given-names>Y.</given-names></name> <name><surname>Lu</surname><given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>Discovering the trading pattern of financial market participants: comparison of two co-clustering methods</article-title>. <source>IEEE Access</source> <volume>6</volume>, <fpage>14431</fpage>&#x2013;<lpage>14438</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2018.2801263</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shin</surname><given-names>H. W.</given-names></name> <name><surname>Sohn</surname><given-names>S. Y.</given-names></name></person-group> (<year>2004</year>). <article-title>Segmentation of stock trading customers according to potential value</article-title>. <source>Expert Syst. Appl.</source> <volume>27</volume>, <fpage>27</fpage>&#x2013;<lpage>33</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.eswa.2003.12.002</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Somanathan</surname><given-names>T. V.</given-names></name> <name><surname>Nageswaran</surname><given-names>V. A.</given-names></name></person-group> (<year>2015</year>). <source>The economics of derivatives</source>: <publisher-name>Cambridge University Press</publisher-name>.</citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Subeesh</surname><given-names>V. K.</given-names></name> <name><surname>Liya</surname><given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>Systematic literature review of top 10 publications in the derivatives market</article-title>. <source>Int. J. Sci. Res.</source> <volume>13</volume>, <fpage>1907</fpage>&#x2013;<lpage>1912</lpage>. doi: <pub-id pub-id-type="doi">10.21275/SR24721123050</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thompson</surname><given-names>J. R.</given-names></name> <name><surname>Feng</surname><given-names>L.</given-names></name> <name><surname>Reesor</surname><given-names>R. M.</given-names></name> <name><surname>Grace</surname><given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>Know your clients&#x2019; behaviours: a cluster analysis of financial transactions</article-title>. <source>J. Risk Financial Manag.</source> <volume>14</volume>:<fpage>50</fpage>. doi: <pub-id pub-id-type="doi">10.3390/jrfm14020050</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vlahavas</surname><given-names>G.</given-names></name> <name><surname>Karasavvas</surname><given-names>K.</given-names></name> <name><surname>Vakali</surname><given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>Unsupervised clustering of bitcoin transactions</article-title>. <source>Financ. Innov.</source> <volume>10</volume>:<fpage>25</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s40854-023-00525-y</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>G.</given-names></name> <name><surname>Nie</surname><given-names>G.</given-names></name> <name><surname>Zhang</surname><given-names>P.</given-names></name> <name><surname>Shi</surname><given-names>Y.</given-names></name></person-group>, (<year>2009</year>). Personal financial market segmentation based on clustering ensembles. in 2009 WRI world congress on computer science and information engineering: IEEE, pp. 694&#x2013;698.</citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wood</surname><given-names>R.</given-names></name> <name><surname>Zaichkowsky</surname><given-names>J. L.</given-names></name></person-group> (<year>2004</year>). <article-title>Attitudes and trading behavior of stock market investors: a segmentation approach</article-title>. <source>J. Behav. Finance</source> <volume>5</volume>, <fpage>170</fpage>&#x2013;<lpage>179</lpage>. doi: <pub-id pub-id-type="doi">10.1207/s15427579jpfm0503_5</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yuen</surname><given-names>K. H.</given-names></name></person-group> (<year>2013</year>). <article-title>The investment preferences and behaviour of small investors in derivatives markets: a survey on derivative investments in Hong Kong</article-title>. <source>J. Emerg. Issues Econ. Finance Bank.</source></citation></ref>
</ref-list>
</back>
</article>