<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Phys.</journal-id>
<journal-title>Frontiers in Physics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Phys.</abbrev-journal-title>
<issn pub-type="epub">2296-424X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1612829</article-id>
<article-id pub-id-type="doi">10.3389/fphy.2025.1612829</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Assessing carbon footprint and performance of the Kalman filter track fitter in the CBM experiment</article-title>
<alt-title alt-title-type="left-running-head">Kozlov and Kisel</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphy.2025.1612829">10.3389/fphy.2025.1612829</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kozlov</surname>
<given-names>Grigory</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2986116/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kisel</surname>
<given-names>Ivan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2852798/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Faculty of Computer Science and Mathematics</institution>, <institution>Goethe University Frankfurt</institution>, <addr-line>Frankfurt am Main</addr-line>, <country>Germany</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Computer Sciences and AI Systems Research Area</institution>, <institution>FIAS, Frankfurt Institute for Advanced Studies</institution>, <addr-line>Frankfurt am Main</addr-line>, <country>Germany</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Technology Research Topic</institution>, <institution>HFHF, Helmholtz Research Academy Hesse</institution>, <addr-line>Frankfurt am Main</addr-line>, <country>Germany</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>CBM Division</institution>, <institution>GSI, Helmholtz Center for Heavy Ion Research</institution>, <addr-line>Darmstadt</addr-line>, <country>Germany</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2222667/overview">Vasiliki A. Mitsou</ext-link>, University of Valencia, Spain</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1702513/overview">Zafar Wazir</ext-link>, Ghazi University, Pakistan</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2915098/overview">Michel Hern&#xe1;ndez Villanueva</ext-link>, Brookhaven National Laboratory (DOE), United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Grigory Kozlov, <email>g.kozlov@gsi.de</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>02</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="ecorrected">
<day>14</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>13</volume>
<elocation-id>1612829</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Kozlov and Kisel.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Kozlov and Kisel</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The CBM experiment at FAIR (GSI, Germany) is among the most significant upcoming projects in heavy-ion physics. It is designed to investigate the properties of dense baryonic matter under extreme conditions. A key feature of the experiment is the high interaction rate, reaching up to <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> collisions per second, resulting in the production of substantial volumes of experimental data that must be processed and analyzed in real time. To meet this computational challenge, the CBM experiment employs a high-performance tracking algorithm based on the Cellular Automaton for track finding and the Kalman Filter for track fitting. The algorithm is designed for efficient parallel execution on modern multicore and GPU architectures. We evaluate the performance and energy efficiency of the Kalman Filter-based fitting algorithm on both CPUs and GPUs. The GPU implementation demonstrates up to a threefold improvement in energy efficiency, resulting in a proportional reduction in power consumption and associated CO<sub>2</sub> emissions during data processing. These results highlight the significance of energy-efficient computing in high-rate heavy-ion experiments. The analysis provides a quantitative estimate of the carbon footprint associated with track reconstruction and demonstrates how hardware choices influence overall emissions in large-scale data processing workflows.</p>
</abstract>
<kwd-group>
<kwd>Kalman filter</kwd>
<kwd>heavy-ion</kwd>
<kwd>CBM experiment</kwd>
<kwd>energy efficiency</kwd>
<kwd>carbon footprint</kwd>
<kwd>parallel computing</kwd>
<kwd>track fitting</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>High-Energy and Astroparticle Physics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Heavy-ion physics central a pivotal role in advancing our understanding of the fundamental properties of matter under extreme conditions. By studying collisions of heavy ions at high energies, researchers probe the behavior of quark-gluon plasma, a state of matter believed to have existed just moments after the Big Bang. These experiments help uncover the forces and interactions between the most basic constituents of matter, providing insight into the early evolution of the universe and guiding our understanding of phase transitions in nuclear matter at unprecedented energy densities.</p>
<p>Experimental facilities, such as particle accelerators and complex detector systems, provide unique opportunities to study the properties of elementary particles and their interactions at extreme energies, conditions that cannot be replicated in low-energy laboratory environments.</p>
<p>Research on particle collisions began with relatively simple tasks that could be handled manually, such as reconstructing particle trajectories in bubble chambers. However, as knowledge and understanding of the complex aspects of matter and interactions grew, the need for more in-depth analysis and investigation of increasingly intricate systems arose. As the volume of experimental data expanded and the complexity of the questions to be addressed increased, the demands on data processing and analysis methods grew significantly.</p>
<p>Modern research in heavy-ion physics is associated with the need to process vast volumes of data generated from high-energy nuclear collisions. This requires not only high-performance computing systems but also advanced analysis methods and algorithms capable of efficiently processing data and extracting meaningful information. As the complexity of the tasks and the volume of data increase, the demands for speed and quality of computational processing become critically important for the successful execution of research and the achievement of new scientific discoveries.</p>
<p>The results of modern experiments in heavy-ion physics can no longer be efficiently processed on individual computers and require the resources of large computational centers. Data analysis necessitates dozens or even hundreds of high-performance processors (CPUs) and graphics accelerators (GPUs), integrated into computing clusters and equipped with powerful cooling systems. These computational systems are major consumers of electrical energy.</p>
<p>On the other hand, the issue of global warming presents the scientific community and industry with the significant challenge of conserving energy and reducing greenhouse gas emissions. The growing energy demands associated with large-scale computing systems are in conflict with the need to reduce the carbon footprint. In the context of global efforts to minimize climate impact, it is essential to explore and implement innovative approaches to improving the energy efficiency of computational infrastructures. In addition, attention must be paid to the efficient use of computational resources, both in terms of task distribution and optimization of the programs and algorithms themselves.</p>
<p>One of the key experiments in heavy-ion physics is the CBM (Compressed Baryonic Matter) [<xref ref-type="bibr" rid="B1">1</xref>] experiment, planned to take place at the FAIR accelerator (Facility for Antiproton and Ion Research). Its primary objective is to study the properties of strongly interacting matter at high net-baryon densities, aiming to explore the QCD phase diagram and the possible existence of a first-order phase transition and a critical point.</p>
<p>New experiments at FAIR are being developed with the latest trends in energy efficiency in mind. As part of this initiative, the Green IT Cube [<xref ref-type="bibr" rid="B2">2</xref>], an innovative computing infrastructure, was built specifically to minimize energy consumption and reduce the carbon footprint. Furthermore, the Goethe HLR (now upgraded to Goethe NHR) [<xref ref-type="bibr" rid="B3">3</xref>], a high-performance computing cluster, is available to support research projects. In the latest Green500 ranking from June 2024 [<xref ref-type="bibr" rid="B4">4</xref>], Goethe NHR holds the 17th position globally in terms of energy efficiency.</p>
<p>This work is dedicated to the study of the energy efficiency of an algorithm that plays a crucial role in particle collision reconstruction, the Kalman filter-based trajectory fitting algorithm [<xref ref-type="bibr" rid="B5">5</xref>]. It is interesting to note that this optimized algorithm has reduced the runtime of the original scalar algorithm by a significant factor of 120.000. Currently, this algorithm is actively used in large-scale heavy-ion physics experiments, such as ALICE at CERN and STAR at BNL, as well as in the track searching and analysis procedures of the future CBM experiment. The high efficiency of the algorithm for track fitting has previously been studied in Kisel and for CBM Collaboration [<xref ref-type="bibr" rid="B6">6</xref>], also investigating the scaling of the number of tracks fitted per time for different computing architectures.</p>
</sec>
<sec id="s2">
<title>2 Research</title>
<sec id="s2-1">
<title>2.1 Algorithm</title>
<p>Charged particle trajectories reconstruction is one of the most important and complex stages of experimental data processing in heavy-ion physics. The reconstruction result is a set of measurements (hits) that form a curve line that corresponds to the track. At the same time, the physical analysis is based not on a set of hits, but on the trajectory parameters within the accepted track model (<xref ref-type="fig" rid="F1">Figure 1a</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Illustration of the concept of track fitting <bold>(a)</bold> and an example of a straight track fitting <bold>(b)</bold>. <bold>(a)</bold> Illustration of the concept of track fitting in a detector with forward geometry. <bold>(b)</bold> An example of a sequential workflow for fitting a simple straight [<xref ref-type="bibr" rid="B7">7</xref>].</p>
</caption>
<graphic xlink:href="fphy-13-1612829-g001.tif">
<alt-text content-type="machine-generated">Diagram labeled &#x201c;a&#x201d; depicts detecting stations in green with a reconstructed track and fitted track in blue and black lines. Diagram labeled &#x201c;b&#x201d; illustrates a state vector of track parameters with steps including initialization, update, and propagation, featuring error bars, hit precision, and optimal estimation.</alt-text>
</graphic>
</fig>
<p>The process of parameterizing the particle&#x2019;s trajectory over a set of discrete measurements is called fitting. Depending on the specific task and the track model used, various algorithms can carry fitting from the simplest (approximation by a line, parabola, circle) to complex multistage algorithms, such as the Kalman Filter (KF). At present, KF is the most popular and frequently used recursive algorithm for estimating track parameters which makes it possible to efficiently calculate the track state vector even in the case of a small number of erroneously attached hits, as well as to make a decision about the correctness of finding the tracks as a whole.</p>
<p>The basic element of the algorithm is the state vector, which includes the track parameters determined by the selected track model. The covariance matrix corresponds to uncertainties and correlations related to the state vector.</p>
<p>After initialization of the parameters and elements of the covariance matrix, the further operation of the algorithm consists of subsequent extrapolation and filtering of the state vector. The extrapolation process is the transport of the state vector to the point of the next measurement, i.e., to the next detector station. Filtering or updating allows us to correct the state vector taking into account the added measurement and recalculate the covariance matrix, considering the predicted and real parameters of the trajectory at the current point (<xref ref-type="fig" rid="F1">Figure 1b</xref>).</p>
<p>The KF track fitter under consideration makes it possible to fit three-dimensional tracks, taking into account additional physical effects that determine the trajectory of a charged particle, such as multiple scattering and a magnetic field. Thus, the algorithm allows one to parameterize a smooth line, which corresponds to the measured particle trajectory.</p>
<p>The KF offers a wide range of possibilities for the parallelization of computations, both on the data level and on the task level. The set of operations performed is the same for each track, but does not depend on third-party cross conditions or the results of processing other tracks. Computing scalability is limited only by the capabilities of the hardware.</p>
<p>One of the key requirements of the CBM experiment is the real-time data processing. This step is essential due to the vast volumes of raw data, which cannot be stored, and the filtering of interesting events requires their full reconstruction. The expected heavy-ion collision rate reaches 10 MHz, with preliminary estimates indicating up to 1,000 particles per event [<xref ref-type="bibr" rid="B8">8</xref>]. Therefore, to successfully carry out the experiment, the speed of the algorithms and the available computational power must be sufficient to reconstruct and fit up to <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> particle trajectories per second.</p>
<p>Kalman Filter plays an important role in the process of track reconstruction and parameterization. Therefore, the speed of the fit has a significant impact on the overall event processing time, while its energy efficiency directly affects the overall energy consumption of the computations.</p>
</sec>
<sec id="s2-2">
<title>2.2 Setup and input data</title>
<p>The SIMD KF Track Fitter package was developed as a benchmark for evaluating the speed of the upgraded KF when using various computing equipment. While the standard KF requires double-precision calculations, the optimized algorithm [<xref ref-type="bibr" rid="B9">9</xref>] shows stable and correct results when working in the single-precision mode. This allows us to take full advantage of the built-in SIMD capabilities of modern CPUs. At the same time, the mutual independence of individual tasks provides almost unlimited scaling of calculations, effectively using the maximum available CPU or GPU threads.</p>
<p>SIMD intrinsics utilization is provided by using special header files or the Vector classes (Vc) [<xref ref-type="bibr" rid="B10">10</xref>] package. Any of the options allows one to quickly switch between different versions of instructions and perform calculations on registers of any length available on the device. Parallelization on the CPU is organized using the OpenMP API and Pthread to provide CPU affinity. Finally, GPU computing is done by using the OpenCL framework.</p>
<p>The test configuration corresponds to the STS detector [<xref ref-type="bibr" rid="B8">8</xref>] of the CBM experiment. The setup includes 7 planes of detecting stations installed one after another to operate in the fixed target mode. The test data is a set of individual tracks selected from central Au&#x2b;Au collisions simulated with CbmRoot at the energy of 25 GeV per nucleon. A suitable track must be reconstructable and have one hit on each of the detecting stations.</p>
<p>The benchmark evaluates only the time of calculations related directly to the track fitting procedure. Time and overhead costs in preparing input data and saving results are not taken into account. Measuring the speed of computing on the CPU using OpenMP is carried out using the standard ROOT class TStopwatch, which makes it possible to get the difference in time between the start and end of calculations. The total execution time of the task is measured, which corresponds to the completion of the slowest thread. It is worth noting that the test data is prepared in such a way that each thread receives a full and equal load. When computing on the GPU, the OpenCL framework profiling methods are used, which allow us to get the start and end times of kernel calculations.</p>
</sec>
<sec id="s2-3">
<title>2.3 Computational resources</title>
<p>The study was conducted on the hybrid computing cluster Goethe NHR at Goethe University in Frankfurt, which is involved in the preparation work for the CBM experiment. Goethe NHR boasts a very good data center energy efficiency ratio, with a Power Usage Effectiveness (<inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the ratio of total energy consumption to the energy consumption of the computational resources) of 1.076 at the time of the research, compared to traditional values for similarly sized data centers, which are typically around &#x223c;1.56 [<xref ref-type="bibr" rid="B11">11</xref>]. Later, Goethe-NHR computing equipment was transferred to the GSI Green IT Cube with even higher energy efficiency PUE&#x3c;1.07. Additionally, when calculating energy efficiency, the carbon intensity factor <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> was considered, which for Germany is approximately 380 <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mi>k</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> according to data from <ext-link ext-link-type="uri" xlink:href="http://Statista.com">Statista.com</ext-link> [<xref ref-type="bibr" rid="B12">12</xref>].</p>
<p>As part of the study, benchmarking was performed on both CPUs and GPUs. The CPU benchmarks were conducted on paired AMD EPYC 7742 processors, built on the Zen 2 architecture, each with 64 cores (128 threads) and a TDP (Thermal Design Power) of approximately 3.52 W per core. The GPU benchmarks were run on AMD MI210 graphics accelerators, which have a TDP of 300 W.</p>
</sec>
<sec id="s2-4">
<title>2.4 Methodology</title>
<p>As the foundation for our study, we selected the methodology for assessing the carbon footprint and energy efficiency of computational tasks presented in the article <italic>Green Algorithms</italic>: Quantifying the Carbon Emissions of Computation [<xref ref-type="bibr" rid="B13">13</xref>]. This methodology enables the calculation of the carbon footprint of computational operations based on parameters such as task execution time, the number of computational cores used (CPU or GPU), the amount of memory involved, and the energy efficiency of the data center where the computations are performed. The method offers a simple and universal approach to estimating the carbon emissions caused by computations, making it applicable to a wide range of tasks and hardware architectures.</p>
<p>For the calculation of energy consumption, the following formula is used:<disp-formula id="e1">
<mml:math id="m6">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>0.001</mml:mn>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x2014; energy consumption (kWh),</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x2014; task execution time (hours),</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x2014; number of computational cores,</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x2014; power consumed by one computational core (W),</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x2014; core utilization factor (from 0 to 1),</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x2014; amount of memory used (GB),</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x2014; power consumed by 1 gigabyte of memory (W),</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf21">
<mml:math id="m22">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x2014; Power Usage Effectiveness of the data center.</p>
</list-item>
</list>
</p>
<p>After calculating the energy consumption, the carbon footprint of the task is determined. For this, the carbon intensity factor <inline-formula id="inf22">
<mml:math id="m23">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which depends on the location and the methods of energy production, is used. The total carbon footprint <inline-formula id="inf23">
<mml:math id="m24">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (in grams of <inline-formula id="inf24">
<mml:math id="m25">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>-equivalent) is calculated using the following formula:<disp-formula id="e2">
<mml:math id="m26">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>This methodology has several strengths. First, it is universal and applicable to many types of computations, from local servers to cloud solutions, making it useful for a wide range of users. Second, it accounts for important parameters, such as the energy efficiency of data centers and the carbon intensity of energy sources, allowing accurate and context-specific calculations.</p>
<p>When using this method for energy efficiency analysis, it is important to keep in mind certain assumptions. Energy consumption during CPU computations is calculated as the product of the number of cores used and the TDP of each core, in other words, <inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from <xref ref-type="disp-formula" rid="e1">Equation 1</xref> is taken as <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>D</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">core</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">cores</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. This approach does not account for modern processor technologies that optimize core performance based on overall device load, which can lead to a slight underestimation of energy consumption when many cores are idle. Additionally, hyper-threading is not considered, and its energy impact can vary significantly depending on the task. On the other hand, if we assume that the processor is already using all available cores and has reached its maximum TDP, adding virtual threads should not affect energy consumption.</p>
<p>As for GPUs, the method assumes that the full TDP is used in the calculations, regardless of the device&#x2019;s actual load. Consequently, the total energy consumption may be overestimated, except in cases where the GPU is operating under maximum load.</p>
<p>In <xref ref-type="fig" rid="F2">Figure 2</xref>, the graphs illustrate the dependence of energy consumption on the utilization levels of both the AMD EPYC 7742 CPU and AMD MI210 GPU, used in this study. To evaluate the CPU power consumption characteristics (<xref ref-type="fig" rid="F2">Figure 2a</xref>), we used both a theoretical estimate based on the number of cores (blue) used and real-world power consumption measurements (red) taken using a 1-phase power analyzer ZES Zimmer LMG95, which collects server power consumption information in real time. The results considered were obtained as the difference between the power consumption under load and the power consumption in idle mode.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Power consumption of the <bold>(a)</bold> CPUs (2x AMD EPYC 7742) and <bold>(b)</bold> GPU (AMD MI210) depending on the model of the device utilization.</p>
</caption>
<graphic xlink:href="fphy-13-1612829-g002.tif">
<alt-text content-type="machine-generated">Two graphs compare power consumption with the number of threads. Graph a shows measured power rising non-linearly and estimation as a linear increase. Graph b shows measured power reaching saturation while estimated power is constant. Both axes are labeled with power (W) and threads.</alt-text>
</graphic>
</fig>
<p>The actual power consumption of the CPU increases non-linearly with higher workloads. The observed behavior is largely determined by the specific characteristics of the processor and the server configuration. In the hardware setup under consideration, the power consumption curve for each CPU exhibits a distinctly convex structure, with a noticeable decline in the incremental power increase after utilizing half of the available cores (<xref ref-type="fig" rid="F2">Figure 2</xref>). At the same time, performance gains remain at a high level, as shown in <xref ref-type="fig" rid="F3">Figure 3a</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>KF Track Fitter scalability measured on the <bold>(a)</bold> CPUs (2x AMD EPYC 7742) and <bold>(b)</bold> GPU (AMD MI210).</p>
</caption>
<graphic xlink:href="fphy-13-1612829-g003.tif">
<alt-text content-type="machine-generated">Two line graphs depict data trends. Graph (a) compares the number of threads to track count, showing a steady increase. Graph (b) contrasts local item size with track count, displaying steep, segmented upward trends. Both graphs use red dots for data points.</alt-text>
</graphic>
</fig>
<p>Since the measurements are conducted with CPU affinity enabled, the observed power consumption is not influenced by the use of virtual threads but is instead a characteristic of the CPU itself. The maximum power consumption of each processor does not reach the specified TDP value. Furthermore, it is worth mentioning that power consumption under low workloads is approximately 50 W higher than expected. This discrepancy is likely attributable to server settings that configure the CPUs and cooling system to maintain minimal energy usage during idle states.</p>
<p>GPU energy consumption was monitored directly during algorithm execution using the <italic>rocm-smi</italic> utility, part of the AMD ROCm software platform. The graph shown in <xref ref-type="fig" rid="F2">Figure 2b</xref> illustrates the dependence of GPU energy consumption during computations on the selected local item size, which corresponds to the device&#x2019;s load level. Even when using only one thread per block on the GPU, the device&#x2019;s power consumption approaches almost half of its full TDP, amounting to 136 W. Idle unused threads still consume energy despite the absence of actual computations. Peak power consumption is reached at an incomplete but sufficiently high level of device utilization and eventually plateaus.</p>
<p>The authors of Green Algorithm offer an energy efficiency assessment of algorithms through either a dedicated website or a console application, which runs on a server and directly interacts with the task scheduler. This approach is primarily aimed at analyzing large-scale tasks with execution times measured in hours. In contrast, the KF Track Fitter is an ultra-fast algorithm, where its speed is defined not by the execution time of a single task, but by the number of tasks completed per unit of time. For this reason, modifications were made to the measurement methodology to evaluate the energy consumption for fitting a specific number of tracks. Additionally, due to the algorithm&#x2019;s wide range of parallelization settings, the effect of these configurations on the final performance metrics was assessed. To account for this, a new factor, <inline-formula id="inf27">
<mml:math id="m29">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, is introduced into <xref ref-type="disp-formula" rid="e1">Equation 1</xref>, where <inline-formula id="inf28">
<mml:math id="m30">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of tracks for energy consumption evaluation, and <inline-formula id="inf29">
<mml:math id="m31">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of tracks processed per unit of time.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<title>3 Results and discussion</title>
<p>The Kalman Filter Track Fitter benchmark provides numerous options for launching parallel computations, allowing researchers to explore the algorithm&#x2019;s performance on various hardware configurations. Vectorized computations are available using all major instruction sets of modern Intel processors: scalar (32-bit, 1 floating point variable), SSE (128-bit, 4 floating point variables), AVX2 (256-bit, 8 floating point variables), and AVX512 (512-bit, 16 floating point variables). Each instruction set has its own characteristics in terms of its impact on CPU energy consumption. However, in our study the testing was primarily conducted using the AVX2 instruction set, as it is the most advanced in terms of vectorization efficiency available on the AMD EPYC CPUs.</p>
<p>Parallel computations are implemented using the OpenMP standard. The program is executed sequentially with varying numbers of threads, ranging from 1 up to 2 CPUs <inline-formula id="inf30">
<mml:math id="m32">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 64 cores <inline-formula id="inf31">
<mml:math id="m33">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 2 threads &#x3d; 256 threads. Computational threads are pinned to cores according to the CPU topology, ensuring that the primary and hyper-threading threads of each core are utilized sequentially. Task distribution is done statically with a step size of 1,000 tracks, while the total number of tracks is a multiple of 1,000. This eliminates overhead associated with task distribution in the given context. As a result, we obtain the algorithm&#x2019;s performance, expressed as the number of tracks fitted per unit of time.</p>
<p>The computations on the GPU are implemented using the OpenCL framework, with optimization for local memory utilization for the most frequently accessed variables. The program is executed with various local item size values, ranging from 1 to 256. Using more threads per block is considered impractical, as the optimal block size for AMD GPUs is 64 elements, and the chosen configurations will allow a full assessment of the relevant patterns. Additionally, an excessive number of threads would lead to local memory overflow and a corresponding decrease in computational speed.</p>
<p>The scalability of the fitting algorithm&#x2019;s computation speed is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. The graph of computation speed growth depending on the number of threads used on the CPU (<xref ref-type="fig" rid="F3">Figure 3a</xref>) exhibits a nearly linear shape with some deviations and changes in slope. The independence of threads and the equivalence of computations in each thread result in a proportional increase in overall performance. The slightly convex shape of the histogram segments reflects the characteristics of the processor&#x2019;s power supply depending on the load.</p>
<p>At low loads, with up to 64 threads in use, we can clearly see the effect of hyperthreading in <xref ref-type="fig" rid="F3">Figure 3a</xref>. This is reflected in the ladder-like structure of the graph, in which the performance gain for every second thread that uses a virtual core is no more than 20%&#x2013;30%. At higher loads, some artifacts caused by CPU power optimization are noticeable, which also affects the efficiency of hyperthreading, making the results difficult to interpret.</p>
<p>The utilization of AVX2 instructions for data-level parallelization ensures high performance of the algorithm by fully populating SIMD vectors with 8 single-precision floating-point elements. The resulting track fitting speed in the context of this study varies from 10.5 <inline-formula id="inf32">
<mml:math id="m34">
<mml:mrow>
<mml:mtext>tracks</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mtext>s</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> when using a single core, to 1071.3 <inline-formula id="inf33">
<mml:math id="m35">
<mml:mrow>
<mml:mtext>tracks</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mtext>s</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> at maximum CPU load.</p>
<p>The AMD MI210 graphics card features 104 compute units (CUs), each supporting 64 threads. This GPU structure defines the scalability characteristics of parallel computations (<xref ref-type="fig" rid="F3">Figure 3b</xref>). The histogram clearly shows a stepwise pattern with increments of 64 elements. Using an incomplete set of threads within a CU results in idle computations and a proportional decrease in the overall performance of the algorithm.</p>
<p>The minimum computation speed is 48.4 <inline-formula id="inf34">
<mml:math id="m36">
<mml:mrow>
<mml:mtext>tracks</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mtext>s</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> with an energy consumption of approximately 135 Wh, which significantly surpasses the CPU&#x2019;s performance under similar load conditions due to the structural differences between the devices. As the number of active threads increases, performance grows at a rate of about 48 <inline-formula id="inf35">
<mml:math id="m37">
<mml:mrow>
<mml:mtext>tracks</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mtext>s</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> per thread. Peak performance reaches 2553.5 <inline-formula id="inf36">
<mml:math id="m38">
<mml:mrow>
<mml:mtext>tracks</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mtext>s</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> when utilizing all 64 threads. With further increases in the local item size, the algorithm&#x2019;s performance decreases in proportion to the ratio of active to idle threads. Subsequent performance peaks show similar, though slightly lower, results. As the number of tasks per CU increases, the amount of both global and local memory used grows, leading to minor additional overhead costs.</p>
<p>The final energy consumption values (<xref ref-type="fig" rid="F4">Figure 4</xref>) were calculated for each pair of total energy consumption per unit of time (<xref ref-type="fig" rid="F2">Figure 2</xref>) and computation speed (<xref ref-type="fig" rid="F3">Figure 3</xref>) using <xref ref-type="disp-formula" rid="e1">Equation 1</xref>, with an estimation for fitting <inline-formula id="inf37">
<mml:math id="m39">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> tracks. In addition to the energy spent directly on the calculations, the equation takes into account the memory energy consumption at the rate of <inline-formula id="inf38">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.3725 W/GB.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Energy consumption of the Kalman Filter Track Fitter for processing <inline-formula id="inf39">
<mml:math id="m41">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> tracks, measured on <bold>(a)</bold> CPUs (2x AMD EPYC 7742) and <bold>(b)</bold> GPU (AMD MI210).</p>
</caption>
<graphic xlink:href="fphy-13-1612829-g004.tif">
<alt-text content-type="machine-generated">Two graphs labeled &#x201c;a&#x201d; and &#x201c;b&#x201d; compare energy consumption (in watt-hours) to the number of threads. Both graphs show real energy consumption as red squares and estimated energy consumption as blue dots. Graph &#x201c;a&#x201d; shows a decrease in energy consumption as threads increase, leveling off after 100 threads. Graph &#x201c;b&#x201d; shows a similar trend, but with more fluctuations in the estimates as threads increase.</alt-text>
</graphic>
</fig>
<p>The Kalman Filter track fitter does not require significant amounts of additional memory for calculations and storing intermediate results. The main memory consumption is for storing information about the detector geometry, as well as directly processed data: tracks and hits they consist of.</p>
<p>In the benchmark under consideration, information about each of the 7 detector stations takes up 156 bytes. In real-world applications, this value can increase to several tens of KB, mainly due to a more detailed map of the station material. This amount of memory use has almost no effect on overall energy consumption.</p>
<p>The main memory consumption is for tracks that are being fitted, primarily due to their number. In the benchmark, this value is 240 bytes per track. Since the fitting process occurs in batches, the amount of memory used <inline-formula id="inf40">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was determined based on the simultaneous storage of up to <inline-formula id="inf41">
<mml:math id="m43">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> tracks. If this parameter is significantly increased, energy costs for temporary data storage start to outweigh the computation costs, which does not align with the algorithm&#x2019;s usage model. Thus, the results provide insight into the approximate energy cost of track fitting that can be expected in the CBM experiment over the course of 1 s, under conditions of maximum particle interaction rates with high collision centrality, when the largest number of fragments is produced.</p>
<p>It should be noted that many of the conditions considered for the upcoming CBM experiment are estimates based on theoretical models and tend to result in inflated numbers. For this reason, the results of the study cannot be regarded as precise values but rather as approximate estimates of the algorithm&#x2019;s energy consumption. These estimates, however, provide valuable insights into the main trends and allow for a rough assessment of the algorithm&#x2019;s energy efficiency.</p>
<p>According to theoretical estimates (<xref ref-type="fig" rid="F2">Figure 2a</xref>, blue markers), fitting <inline-formula id="inf42">
<mml:math id="m44">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> particle trajectories using the Kalman filter on modern CPUs required between 0.83 and 1.25 Wh of electricity (<xref ref-type="fig" rid="F4">Figure 4a</xref>, blue markers). The wavelike structure of the histogram is caused by the uneven growth in computation speed with linear increases in energy consumption.</p>
<p>Actual power consumption measurements (<xref ref-type="fig" rid="F4">Figure 4a</xref>, red markers) show clear differences from the estimates in areas of high or low CPU utilization, whereas they are rather close to the estimates for medium CPU utilization. Computations on a small number of threads lead to relatively low power efficiency, which is an obvious effect of the jump and a sharp further growth of power consumption with a more uniform increase in the speed of calculations. At the same time, the power efficiency at maximum CPU load looks better than theoretical. Fitting <inline-formula id="inf43">
<mml:math id="m45">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> tracks in this case requires about 1.00 Wh of electricity.</p>
<p>Now, knowing the energy cost for fitting <inline-formula id="inf44">
<mml:math id="m46">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> tracks&#x2014;representing the hypothetical peak track output per second&#x2014;we can calculate the carbon footprint for 1 day of algorithm operation using <xref ref-type="disp-formula" rid="e2">Equation 2</xref>, as outlined in the <italic>Green Algorithms</italic> methodology. The choice of a 24-h time interval is made for clarity and ease of further extrapolation, as the experiment and data processing will be conducted continuously around the clock.<disp-formula id="equ1">
<mml:math display="block" id="m47">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>24</mml:mn>
<mml:mtext>h</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>60</mml:mn>
<mml:mtext>min</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>60</mml:mn>
<mml:mtext>sec</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>0.0010</mml:mn>
<mml:mtext>kWh</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>380</mml:mn>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mtext>gCO</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mtext>kWh</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>32,832</mml:mn>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mtext>gCO</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Thus, particle trajectory fitting in the CBM experiment using CPUs at maximum load could result in up to 32.8 kg of <inline-formula id="inf45">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> emissions per day. To make this more relatable, these emissions can be expressed in terms of driving distance and carbon sequestration (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Comparison of energy consumption across different computing setups under optimal resource utilization.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Hardware platform</th>
<th align="center">Energy consumption<break/> (Wh per <inline-formula id="inf47">
<mml:math id="m50">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>tracks)</th>
<th align="center">
<inline-formula id="inf46">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> emission<break/> <inline-formula id="inf48"> <mml:math id="m51"> <mml:mrow> <mml:msub> <mml:mrow> <mml:mtext>kgCO</mml:mtext> </mml:mrow> <mml:mrow> <mml:mn>2</mml:mn> </mml:mrow> </mml:msub> </mml:mrow> </mml:math> </inline-formula>
</th>
<th align="center">Driving distance<break/> (km)</th>
<th align="center">Carbon sequestration<break/> (Tree-mounts)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">CPU (AMD EPYC 7742)</td>
<td align="center">1.00</td>
<td align="center">32.8</td>
<td align="center">321</td>
<td align="center">35.8</td>
</tr>
<tr>
<td align="left">GPU (AMD MI210)</td>
<td align="center">0.35</td>
<td align="center">11.5</td>
<td align="center">112</td>
<td align="center">11.5</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>According to open data from the European Environment Agency [<xref ref-type="bibr" rid="B14">14</xref>], the average <inline-formula id="inf49">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> emissions for new cars in Europe are <inline-formula id="inf50">
<mml:math id="m53">
<mml:mrow>
<mml:mn>102.2</mml:mn>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mtext>gCO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mtext>km</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>. Therefore, the <inline-formula id="inf51">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> emissions in the scenario considered would correspond to:<disp-formula id="equ2">
<mml:math id="m55">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>32,832</mml:mn>
<mml:mtext> </mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mtext>gCO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>102.2</mml:mn>
<mml:mtext> </mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mtext>gCO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mtext>km</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2248;</mml:mo>
<mml:mn>321</mml:mn>
<mml:mtext> </mml:mtext>
<mml:mtext>km</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Consistent with studies [<xref ref-type="bibr" rid="B15">15</xref>], a tree absorbs 10&#x2014;12 <inline-formula id="inf52">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>kgCO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> per year, depending on growth conditions, or approximately <inline-formula id="inf53">
<mml:math id="m57">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mtext>kg/month</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>. Thus, the daily <inline-formula id="inf54">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> emissions from the fitting of the experimental data would be equivalent to about 3 tree-year. These comparisons give a clearer sense of the environmental impact of the computational tasks involved in the CBM experiment.</p>
<p>By comparison, processing <inline-formula id="inf55">
<mml:math id="m59">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> tracks at minimum CPU load requires approximately 17 times more electricity, resulting in a corresponding increase in carbon dioxide production.</p>
<p>As a means of self-validation, we also compared the calculated results with the output data from the online calculator <italic>Green Algorithms</italic> (<xref ref-type="fig" rid="F5">Figure 5</xref>), whose operation served as the foundation for developing the research methodology. The calculations performed by the online calculator were based on the use of 1,195 CPU cores over 24 h. This setup is expected to enable the fitting of <inline-formula id="inf56">
<mml:math id="m60">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> tracks per second, according to previously obtained estimates of the algorithm&#x2019;s performance (<xref ref-type="fig" rid="F3">Figure 3a</xref>). It should also be noted that the measured peak energy consumption of the system under consideration was 20% lower than the maximum values used by the authors of <italic>Green Algorithms</italic> by default.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The estimated results for electricity consumption and carbon footprint of fitting <inline-formula id="inf57">
<mml:math id="m61">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> tracks per second in real time over 24 h, determined using the <italic>Green Algorithms</italic> online calculator.</p>
</caption>
<graphic xlink:href="fphy-13-1612829-g005.tif">
<alt-text content-type="machine-generated">Infographic displaying environmental impact metrics. It shows 36.73 kg CO2e carbon footprint, 108.46 kWh energy needed, carbon sequestration equivalent to 3.34 tree-years, 209.89 km traveled in a passenger car, and 73 percent of a Paris-London flight.</alt-text>
</graphic>
</fig>
<p>Daily electricity consumption, according to our calculations, amounted to 86.4 kWh, which is approximately 20% lower than the result provided by the online calculator and therefore fully aligns with the expected values. The difference in the mass of the emitted <inline-formula id="inf58">
<mml:math id="m62">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is only about 10%. This discrepancy arises from the use of different values for the carbon intensity factor in the calculations. Specifically, <italic>Green Algorithms</italic> uses data from 2020 (338.66 <inline-formula id="inf59">
<mml:math id="m63">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mi>k</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), while our research is based on more recent figures from 2023 (380 <inline-formula id="inf60">
<mml:math id="m64">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mi>k</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>). Accounting for this difference reconciles the results. An even more pronounced discrepancy in the driving distance metric reflects not only the influence of prior carbon emission calculations, but also a significant difference in the average <inline-formula id="inf61">
<mml:math id="m65">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> emissions per 100 km for vehicles: 175 <inline-formula id="inf62">
<mml:math id="m66">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mi>k</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (2019) in <italic>Green Algorithms</italic> compared to 102.2 <inline-formula id="inf63">
<mml:math id="m67">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mi>k</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (2024) in our study.</p>
<p>Thus, it can be concluded that the results of our study align with the stated methodology and correlate well with the estimates provided by the <italic>Green Algorithms</italic> online calculator. However, it is equally important to note that indirect estimations are highly dependent on the coefficients used, which may vary over time or differ depending on the source. Such estimates can be utilized to enhance the interpretability of the results, but they are not sufficiently precise on their own.</p>
<p>The dependence of GPU energy consumption on the device load is much more pronounced (<xref ref-type="fig" rid="F4">Figure 4b</xref>). We will examine two extreme cases in terms of carbon footprint and compare the obtained results with the data from the CPU.</p>
<p>The worst energy efficiency occurs when only one thread per compute unit is used. Low computational performance combined with high energy consumption results in a requirement of approximately 8.34 Wh to process <inline-formula id="inf64">
<mml:math id="m68">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> tracks. These computations result in up to 273.8 <inline-formula id="inf65">
<mml:math id="m69">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>kgCO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> per day. This is equivalent to driving a car for a distance of 2,679 km. Compensation for such emissions would require around 274 tree-months of <inline-formula id="inf66">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> sequestration.</p>
<p>Optimal GPU energy efficiency is largely determined by the proper utilization of computational resources to achieve maximum fitting speed. This corresponds to configurations where all threads are utilized, that is, when the local item size is a multiple of 64. The peak efficiency is similar across settings, but the best value is achieved with a local item size of 64, resulting in an energy consumption of 0.35 Wh per <inline-formula id="inf67">
<mml:math id="m71">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> tracks.</p>
<p>Converting this result similarly to the previously obtained data, we arrive at the following values. Daily <inline-formula id="inf68">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> emissions amount to approximately 11.5 kg, which is 3 times less than in the case of CPU usage. This is equivalent to a trip by car of 112.4 km. Such environmental impact would be compensated for by approximately 12.6 tree-months of <inline-formula id="inf69">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> sequestration.</p>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>In this work, an analysis of the efficiency of the Kalman Filter-based fitting algorithm, which plays a key role in particle trajectory reconstruction in heavy-ion physics experiments such as the CBM experiment at FAIR, was conducted. The study included an assessment of computational resources, energy consumption, and carbon footprint when executing the algorithm on modern processors and graphics accelerators. The results show that the algorithm, optimized for SIMD instructions and multithreading, provides high performance and efficiency in the reconstruction and analysis of particle trajectories.</p>
<p>Further analysis reveals that CPU energy efficiency improves with increasing thread count, but power management mechanisms lead to nonuniform efficiency gains. At low CPU loads, baseline power consumption reduces efficiency, whereas at peak loads, real energy usage remains below theoretical estimates. GPU efficiency depends on optimal resource utilization&#x2014;under low occupancy, energy costs rise sharply, while full utilization achieves up to three times lower energy consumption per fitted track compared to the CPU.</p>
<p>This reduction in energy consumption leads to a threefold decrease in <inline-formula id="inf70">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CO</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> emissions and, consequently, in the amount of carbon sequestration needed. Therefore, optimizing the algorithm for GPU execution not only enhances computational efficiency but also significantly reduces the carbon footprint of experimental data processing in heavy-ion physics.</p>
<p>The data obtained highlight the importance of further optimizing algorithms and computational systems to achieve a balance between performance and energy efficiency, especially in the context of increasing data volumes and heightened demands for computational power. In the face of global efforts to reduce carbon footprints, the development of &#x201c;green&#x201d; computing algorithms and infrastructures is becoming an integral part of the modern scientific process, particularly in the field of heavy-ion physics.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://drive.google.com/file/d/1fhft9Idc-ixYgQoIcoGlsJa5uuDbxJtm/view?usp%26equals;sharinghttps://drive.google.com/file/d/1AM582g4on6AuH6JVJKK1QVhYNvJ6Lyvj/view?usp&#x26;equals;sharing">https://drive.google.com/file/d/1fhft9Idc-ixYgQoIcoGlsJa5uuDbxJtm/view?usp&#x26;equals;sharinghttps://drive.google.com/file/d/1AM582g4on6AuH6JVJKK1QVhYNvJ6Lyvj/view?usp&#x26;equals;sharing</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>GK: Conceptualization, Investigation, Writing &#x2013; review and editing, Validation, Data curation, Writing &#x2013; original draft, Project administration, Formal Analysis, Methodology, Resources, Software, Visualization. IK: Writing &#x2013; review and editing, Formal Analysis, Project administration, Data curation, Conceptualization, Validation.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was partly funded by the German Federal Ministry of Education and Research [grant numbers 05H21VKRC2, 01IS21092, 05P24RF3, 05P24RF7], Germany, and Helmholtz Research Academy Hesse for FAIR (project ID 2.1.4.2.5), Germany.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9">
<title>Correction note</title>
<p>This article has been corrected with minor changes. These changes do not impact the scientific content of the article.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. Generative AI (ChatGPT 4o) was used to check grammar and make minor stylistic adjustments to some parts of the main text and conclusion. AI was not used to generate descriptions of significant parts of the study.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Senger</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Friese</surname>
<given-names>V</given-names>
</name>
</person-group>. <source>CBM progress report 2021. Progress report GSI-2022-00599</source>. <publisher-loc>Darmstadt, Germany</publisher-loc>: <publisher-name>GSI Darmstadt</publisher-name> (<year>2022</year>). <pub-id pub-id-type="doi">10.15120/GSI-2022-00599</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="book">
<collab>GSI Helmholtzzentrum f&#xfc;r Schwerionenforschung</collab>. <article-title>Green IT Cube: supercomputing center for GSI and FAIR. (2023)</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.gsi.de/en/researchaccelerators/research_an_overview/green-it-cube">https://www.gsi.de/en/researchaccelerators/research_an_overview/green-it-cube</ext-link>. (Accessed June 12, 2025)</comment>.</citation>
</ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="book">
<collab>Center for Scientific Computing Goethe University Frankfurt</collab>. <article-title>Goethe&#x2013;NHR: performance and hardware details. (2024)</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://csc.unifrankfurt.de/wiki/doku.phpid=public:service:goethe-hlr">https://csc.unifrankfurt.de/wiki/doku.phpid=public:service:goethe-hlr</ext-link>. (Accessed June 12, 2025)</comment>.</citation>
</ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="book">
<collab>TOP500</collab>. <article-title>Green 500 list, 2024. (2024)</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://top500.org/lists/green500/2024/06/">https://top500.org/lists/green500/2024/06/</ext-link>. (Accessed June 12, 2025)</comment>.</citation>
</ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gorbunov</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Kebschull</surname>
<given-names>U</given-names>
</name>
<name>
<surname>Kisel</surname>
<given-names>I</given-names>
</name>
<name>
<surname>Lindenstruth</surname>
<given-names>V</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname>
<given-names>W</given-names>
</name>
</person-group>. <article-title>Fast SIMDized Kalman filter based track fit</article-title>. <source>Computer Phys Commun</source> (<year>2008</year>) <volume>178</volume>:<fpage>374</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1016/j.cpc.2007.10.001</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kisel</surname>
<given-names>I</given-names>
</name>
</person-group>
<collab>for CBM Collaboration</collab>. <article-title>Event topology reconstruction in the CBM experiment</article-title>. <source>J Phys Conf Ser</source> (<year>2018</year>) <volume>1070</volume>:<fpage>012015</fpage>. <pub-id pub-id-type="doi">10.1088/1742-6596/1070/1/012015</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kozlov</surname>
<given-names>G.</given-names>
</name>
</person-group> <article-title>Cellular automaton based track finder in the STAR experiment (BNL, USA). Doctoralthesis. Frankfurt am Main, Germany: Universit&#xe4;tsbibliothek Johann Christian Senckenberg</article-title> (<year>2022</year>).</citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="book">
<person-group person-group-type="editor">
<name>
<surname>Heuser</surname>
<given-names>J</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Pugatch</surname>
<given-names>V</given-names>
</name>
<name>
<surname>Senger</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>CJ</given-names>
</name>
<name>
<surname>Sturm</surname>
<given-names>C</given-names>
</name>
<name>
<surname>et al.</surname>
</name>
</person-group> editors <source>[GSI report 2013-4] technical Design report for the CBM silicon tracking system (STS)</source>. <publisher-loc>Darmstadt</publisher-loc>: <publisher-name>GSI</publisher-name> (<year>2013</year>).</citation>
</ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Zyzak</surname>
<given-names>M</given-names>
</name>
</person-group>. <source>Online selection of short-lived particles on many-core computer architectures in the CBM experiment at FAIR</source>. <comment>Doctoralthesis</comment>. <publisher-loc>Frankfurt am Main, Germany</publisher-loc>: <publisher-name>Universit&#xe4;tsbibliothek Johann Christian Senckenberg</publisher-name> (<year>2016</year>).</citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kretz</surname>
<given-names>M</given-names>
</name>
</person-group>. <article-title>Extending C&#x2b;&#x2b; for explicit data-parallel programming via SIMD vector types. doctoralthesis</article-title>. <source>Universit&#xe4;tsbibliothek Johann Christian Senckenberg</source> (<year>2015</year>).</citation>
</ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="book">
<collab>Uptime Institute Research Team</collab>. <article-title>Uptime Institute Global Data Center Survey 2024. (2024)</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://datacenter.uptimeinstitute.com/rs/711-RIA-145/images/2024.GlobalDataCenterSurvey.Report.pdf">https://datacenter.uptimeinstitute.com/rs/711-RIA-145/images/2024.GlobalDataCenterSurvey.Report.pdf</ext-link>
</comment>.</citation>
</ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="book">
<collab>Statista</collab>. <article-title>Development of the CO2 emissions factor in theelectricity mix in Germany from 1990 to 2023. (2023)</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.statista.com/statistics/1386327/co2-emissions-factor-electricity-mix-germany/">https://www.statista.com/statistics/1386327/co2-emissions-factor-electricity-mix-germany/</ext-link> (Accessed June 12, 2025)</comment>.</citation>
</ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lannelongue</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Grealey</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Inouye</surname>
<given-names>M</given-names>
</name>
</person-group>. <article-title>Green algorithms: quantifying the carbon footprint of computation</article-title>. <source>Adv Sci</source> (<year>2021</year>) <volume>8</volume>:<fpage>2100707</fpage>. <pub-id pub-id-type="doi">10.1002/advs.202100707</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="book">
<collab>European Environment Agency</collab>. <article-title>Average CO<sub>2</sub> emissions of pools of car manufacturers. (2024)</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.eea.europa.eu/en/analysis/maps-and-charts/data-visualization-55">https://www.eea.europa.eu/en/analysis/maps-and-charts/data-visualization-55</ext-link>. (Accessed June 12, 2025)</comment>.</citation>
</ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Akbari</surname>
<given-names>H</given-names>
</name>
</person-group>. <article-title>Shade trees reduce building energy use and CO<sub>2</sub> emissions from power plants</article-title>. <source>Environ Pollut</source> (<year>2002</year>) <volume>116</volume>:<fpage>S119</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1016/S0269-7491(01)00264-0</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>