<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mater.</journal-id>
<journal-title>Frontiers in Materials</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mater.</abbrev-journal-title>
<issn pub-type="epub">2296-8016</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">851458</article-id>
<article-id pub-id-type="doi">10.3389/fmats.2022.851458</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Materials</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Fast All-Electron Hybrid Functionals and Their Application to Rare-Earth Iron Garnets</article-title>
<alt-title alt-title-type="left-running-head">Redies et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">Fast All-Electron Hybrid Functionals</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Redies</surname>
<given-names>Matthias</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1426449/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Michalicek</surname>
<given-names>Gregor</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1660447/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bouaziz</surname>
<given-names>Juba</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1648420/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Terboven</surname>
<given-names>Christian</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>M&#xfc;ller</surname>
<given-names>Matthias S.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bl&#xfc;gel&#x2009;</surname>
<given-names>Stefan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1154173/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wortmann</surname>
<given-names>Daniel</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1210485/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Peter Gr&#xfc;nberg Institut and Institute for Advanced Simulation</institution>, <institution>Forschungszentrum J&#xfc;lich</institution>, <addr-line>J&#xfc;lich</addr-line>, <country>Germany</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>JARA-CSD</institution>, <addr-line>J&#xfc;lich</addr-line>, <country>Germany</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Physics</institution>, <institution>RWTH Aachen University</institution>, <addr-line>Aachen</addr-line>, <country>Germany</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>IT Center</institution>, <institution>RWTH Aachen University</institution>, <addr-line>Aachen</addr-line>, <country>Germany</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/125847/overview">Zhenyu Li</ext-link>, University of Science and Technology of China, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/860214/overview">Honghui Shang</ext-link>, Institute of Computing Technology (CAS), China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/859381/overview">Mohan Chen</ext-link>, Peking University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/642159/overview">Wei Hu</ext-link>, Lawrence Berkeley National Laboratory, United&#x20;States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Daniel Wortmann, <email>d.wortmann@fz-juelich.de</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Materials Science, a section of the journal Frontiers in Materials</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>03</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>9</volume>
<elocation-id>851458</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>02</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Redies, Michalicek, Bouaziz, Terboven, M&#xfc;ller, Bl&#xfc;gel&#x2009; and Wortmann.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Redies, Michalicek, Bouaziz, Terboven, M&#xfc;ller, Bl&#xfc;gel&#x2009; and Wortmann</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>Virtual materials design requires not only the simulation of a huge number of systems, but also of systems with ever larger sizes and through increasingly accurate models of the electronic structure. These can be provided by density functional theory (DFT) using not only simple local approximations to the unknown exchange and correlation functional, but also more complex approaches such as hybrid functionals, which include some part of Hartree&#x2013;Fock exact exchange. While hybrid functionals allow many properties such as lattice constants, bond lengths, magnetic moments and band gaps, to be calculated with improved accuracy, they require the calculation of a nonlocal potential, resulting in high computational costs, that scale rapidly with the system size. This limits their wide application. Here, we present a new highly-scalable implementation of the nonlocal Hartree-Fock-type potential into FLEUR&#x2014;an all-electron electronic structure code that implements the full-potential linearized augmented plane-wave (FLAPW) method. This implementation enables the use of hybrid functionals for systems with several hundred atoms. By porting this algorithm to GPU accelerators, we can leverage future exascale supercomputers which we demonstrate by reporting scaling results for up to 64 GPUs and up to 12,000 CPU cores for a single <bold>k</bold>-point. As proof of principle, we apply the algorithm to large and complex iron garnet materials (YIG, GdIG, TmIG) that are used in several spintronic applications.</p>
</abstract>
<kwd-group>
<kwd>Density Functional Theory (DFT)</kwd>
<kwd>Rare-Earth Iron Garnets</kwd>
<kwd>High-Performance Computing (HPC)</kwd>
<kwd>PBE0</kwd>
<kwd>Hybrid Functionals</kwd>
<kwd>YIG</kwd>
<kwd>GdIG</kwd>
<kwd>TmIG</kwd>
</kwd-group>
<contract-sponsor id="cn001">Horizon 2020 Framework Programme<named-content content-type="fundref-id">10.13039/100010661</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Materials science aims to understand and predict material properties ever more accurately, so that new sophisticated materials can be discovered to drive innovation in domains that rely on them. While materials science has been around for millennia, it was only at the beginning of the last century that the arrival of quantum mechanics enabled the exact description of the microscopic properties in materials. However, the cost of calculating the exact solution to the Schr&#xf6;dinger equation grows exponentially with the size of the system and is therefore limited to very small systems. Density functional theory (DFT) replaces the 3<italic>N</italic>-dimensional wave function as the central quantity with the 3-dimensional ground-state density and thereby reduces the exponential computational cost to a polynomial one. While DFT is in principle exact, a key ingredient, the so-called exchange-correlation energy, has no known analytical expression. The approximations used for this term determine the accuracy with which material properties can be predicted. While the most commonly used approximations, the local density approximation (LDA) and the generalized gradient approximation (GGA), can predict certain properties with a high precision at a very low computational cost, they fail to predict some essential electronic properties in particular of electronically complex materials (<xref ref-type="bibr" rid="B1">Alberi et&#x20;al., 2018</xref>).</p>
<p>DFT is increasingly being used in the context of high-throughput calculations, where hundreds of thousands of material candidates are screened using automated workflows (<xref ref-type="bibr" rid="B50">Yan et&#x20;al., 2017</xref>; <xref ref-type="bibr" rid="B33">Mounet et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B38">Rosen et&#x20;al., 2019</xref>). However, all of these calculations are limited to material classes and properties for which the underlying exchange-correlation functionals have a good predictive power. In order to enhance these calculations with material classes and properties for which LDA and GGA fail, it is necessary to rely on more accurate methods producing high-quality results. One class of accurate methods are the hybrid exchange-correlation functionals which are particularly suited to predicting electronic properties such as the band gap, the degree of charge localization and the polarization in materials with a stronger electron correlation (<xref ref-type="bibr" rid="B13">Cramer and Truhlar, 2009</xref>; <xref ref-type="bibr" rid="B52">Zhang et&#x20;al., 2011</xref>; <xref ref-type="bibr" rid="B10">Burke, 2012</xref>; <xref ref-type="bibr" rid="B6">Becke, 2014</xref>; <xref ref-type="bibr" rid="B18">Garza and Scuseria, 2016</xref>).</p>
<p>Hybrid exchange-correlation functionals, such as PBE0 (<xref ref-type="bibr" rid="B35">Perdew et&#x20;al., 1996</xref>) or HSE06 (<xref ref-type="bibr" rid="B24">Krukau et&#x20;al., 2006</xref>) functionals, mix a portion of an orbital dependent exact exchange with the electron correlation described by other approximations, such as LDA or GGA (<xref ref-type="bibr" rid="B5">Becke, 1993</xref>). Their reliance on the orbital dependent exact exchange makes them computationally considerably more expensive than LDA or GGA. While an LDA or a GGA calculation grows with the 3rd power of the number of atoms, a hybrid exchange-correlation functional calculation typically grows with the 4th power of the number of atoms. Additionally, the computational cost of a hybrid calculation grows quadratically with the number of <bold>k</bold>-points used to sample the Brillouin zone, whereas for an LDA or a GGA calculation it only grows linearly. This large computational cost has prohibited precise predictions for systems with large unit cells, including a number of interesting material classes such as garnets (<xref ref-type="bibr" rid="B37">Rodic et&#x20;al., 1999</xref>; <xref ref-type="bibr" rid="B34">Nakamoto et&#x20;al., 2017</xref>) or materials of interest for solid-state batteries (<xref ref-type="bibr" rid="B51">Yu et&#x20;al., 2016</xref>).</p>
<p>This article focuses on the implementation of hybrid functionals in the full-potential linearized augmented-plane-wave (FLAPW) (<xref ref-type="bibr" rid="B48">Wimmer et&#x20;al., 1981</xref>) method as it is implemented in the open-source code FLEUR (<xref ref-type="bibr" rid="B16">Fleur, 2021</xref>). Unlike approaches employing the pseudopotential approximation, the FLAPW method treats all electrons explicitly and does not employ any approximations to represent the potential or density. It is therefore well suited for a wide range of systems, including systems containing heavy atoms that have <italic>d</italic>- and <italic>f</italic>-electrons. It is considered to be one of the most accurate DFT methods and has been used as a benchmark for other methods and codes (<xref ref-type="bibr" rid="B28">Lejaeghere et&#x20;al., 2016</xref>). More specifically we focus here on the efficient implementation of the Hartree-Fock type exact exchange based on the bare Coulomb kernel as relevant for the PBE0 functional. Functionals based on the screened Coulomb kernel as HSE06 can be always expressed in terms of the matrix elements of the bare Coulomb kernel subtracted by matrix elements of a smooth function (<xref ref-type="bibr" rid="B40">Schlipf et&#x20;al., 2011</xref>), whose numerical evaluation is not time critical and is not further discussed&#x20;here.</p>
<p>There have been significant advances in bringing hybrid functionals to systems with hundreds or even thousands of atoms in other approaches, such as the projector augmented wave method (PAW) (<xref ref-type="bibr" rid="B4">Barnes et&#x20;al., 2017</xref>; <xref ref-type="bibr" rid="B11">Carnimeo et&#x20;al., 2019</xref>), the s-MTACE METHOD (<xref ref-type="bibr" rid="B31">Mandal et&#x20;al., 2021</xref>), gaussian basis functions (<xref ref-type="bibr" rid="B21">Guidon et&#x20;al., 2008</xref>) and atomic-orbitals method (<xref ref-type="bibr" rid="B22">Hakala and Foster, 2013</xref>; <xref ref-type="bibr" rid="B30">Lin et&#x20;al., 2021</xref>). Even some all-electron methods have demonstrated their capability to calculate large systems with hybrid functionals (<xref ref-type="bibr" rid="B23">Ihrig et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B29">Levchenko et&#x20;al., 2015</xref>). However, hybrid functionals within FLAPW have been constrained to very small systems (<xref ref-type="bibr" rid="B7">Betzinger et&#x20;al., 2010</xref>; <xref ref-type="bibr" rid="B40">Schlipf et&#x20;al., 2011</xref>; <xref ref-type="bibr" rid="B8">Blaha et&#x20;al., 2020</xref>). The work presented here enables FLEUR&#x2019;s hybrid functional implementation to run on the world&#x2019;s most advanced supercomputers and use their immense computational power to investigate these large and interesting systems containing hundreds of atoms. Building on the pioneering work previously done on hybrid functionals in the FLAPW basis and in FLEUR specifically (<xref ref-type="bibr" rid="B7">Betzinger et&#x20;al., 2010</xref>; <xref ref-type="bibr" rid="B40">Schlipf et&#x20;al., 2011</xref>), we analyzed the performance and bottlenecks of this legacy implementation, and explored algorithmic improvements needed to calculate hundreds of atoms with the accuracy that FLAPW and hybrid functionals&#x20;offer.</p>
</sec>
<sec id="s2">
<title>2 Methods</title>
<p>The FLAPW method, implemented by FLEUR, partitions the unit cell of volume &#x3a9; into two kinds of domains. In spherical regions MT<sub>
<italic>a</italic>
</sub> centered around each atomic nucleus, a muffin-tin orbital basis (<xref ref-type="bibr" rid="B2">Andersen and Woolley, 1973</xref>) is used, relying on the products of spherical harmonics and radial functions. In between these spheres, in the so-called interstitial region (IS), a plane-wave basis is used. The resulting LAPW basis functions used to represent the wave functions are<disp-formula id="e1">
<mml:math id="m1">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">k</mml:mi>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mspace width="0.17em"/>
<mml:mi>exp</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">f</mml:mi>
<mml:mspace width="0.17em"/>
<mml:mi mathvariant="bold">r</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="normal">I</mml:mi>
<mml:mi mathvariant="normal">S</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mspace width="0.17em"/>
<mml:msubsup>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2016;</mml:mi>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x2016;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mspace width="0.17em"/>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mo>&#x307;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2016;</mml:mi>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x2016;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">f</mml:mi>
<mml:mspace width="0.17em"/>
<mml:mi mathvariant="bold">r</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>Here <bold>k</bold> and <bold>G</bold> are the Bloch- and reciprocal lattice vectors, while <italic>&#x3c3;</italic> indicates the spin. <inline-formula id="inf1">
<mml:math id="m2">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> and <inline-formula id="inf2">
<mml:math id="m3">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> are coefficients chosen, such that the wave function is smooth and continuous at the boundary between the interstitial region and muffin-tin spheres. <italic>u</italic> and <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mo>&#x307;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> are radial functions, where <italic>u</italic> is the solution to the radial Schr&#xf6;dinger equation for the spherically averaged muffin-tin potential and a fixed energy parameter and <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mo>&#x307;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is its energy derivative. <italic>a</italic> indicates the nucleus, <italic>l</italic> and <italic>m</italic> are the orbital- and magnetic quantum numbers of the spherical harmonic <italic>Y</italic>
<sub>
<italic>lm</italic>
</sub>. <bold>r</bold> denotes the position, while <bold>r</bold>
<sub>
<italic>a</italic>
</sub>&#x2254;<bold>r</bold> &#x2212; <bold>R</bold>
<sub>
<italic>a</italic>
</sub> is the position relative to the center of the muffin-tin sphere and <bold>e</bold>
<sub>
<italic>a</italic>
</sub>&#x2254;<bold>r</bold>
<sub>
<italic>a</italic>
</sub>/norm&#x2009;<bold>r</bold>
<sub>
<italic>a</italic>
</sub> is the unit vector in direction of&#x20;<bold>r</bold>
<sub>
<italic>a</italic>
</sub>.</p>
<p>In order to calculate the Hartree-Fock exact exchange, the Coulomb integral <inline-formula id="inf5">
<mml:math id="m6">
<mml:mo>&#x222c;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x03c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2217;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:math>
</inline-formula>, containing four basis functions needs to be evaluated. It has previously been noted (<xref ref-type="bibr" rid="B9">Boys and Shavitt, 1959</xref>; <xref ref-type="bibr" rid="B42">Shavitt, 1959</xref>), that the product of the basis functions sharing the same argument <inline-formula id="inf6">
<mml:math id="m7">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2217;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf7">
<mml:math id="m8">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2217;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> can be expressed in a more efficient way, since the basis function are already designed to be complete and the set of all products <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2217;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">r</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> makes a linear dependent set. In the case of the LAPW basis, this observation can be exploited by employing the so-called mixed-product basis. The mixed-product basis reduces the basis set for the muffin-tin regions separately from the interstitial region. In the muffin-tin regions, the overlap matrix of the products is calculated and diagonalized. The eigenvectors whose eigenvalues are above a certain threshold <italic>&#x3bd;</italic> then provide a linear-independent representation of the product space. In the interstitial region products of plane-waves are also planes-waves, but with a higher cut-off <inline-formula id="inf9">
<mml:math id="m10">
<mml:msubsup>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>max</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>max</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, where <italic>G</italic>
<sub>max</sub> is the plane-wave cut-off of the LAPW basis. In practice reduced values of <inline-formula id="inf10">
<mml:math id="m11">
<mml:msubsup>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>max</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> have proven to provide accurate results. While this new mixed-product basis is neither continuous nor smooth, it provides a significant reduction in computational effort compared to the naive evaluation of the Coulomb integral. A detailed description of the mixed-product basis can be found in (<xref ref-type="bibr" rid="B17">Friedrich et&#x20;al., 2009</xref>).</p>
<sec id="s2-1">
<title>2.1 Exact Exchange</title>
<p>Using this basis set the nonlocal exact exchange can be expressed as<disp-formula id="e2">
<mml:math id="m12">
<mml:msubsup>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mtext>exact</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2033;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>occ</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munderover>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>BZ</mml:mtext>
</mml:mrow>
</mml:munderover>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mfenced open="&#x27e8;" close="&#x27e9;">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2033;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold">k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold">q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="&#x27e8;" close="&#x27e9;">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2033;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold">k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold">q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(2)</label>
</disp-formula>where <italic>n</italic>, <italic>n</italic>&#x2032; and <italic>n</italic>&#x2032;&#x2032; are band indices of the states <inline-formula id="inf11">
<mml:math id="m13">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>, <inline-formula id="inf12">
<mml:math id="m14">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> and <inline-formula id="inf13">
<mml:math id="m15">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>. <italic>I</italic> and <italic>J</italic> are indices enumerating the mixed-product basis and <inline-formula id="inf14">
<mml:math id="m16">
<mml:mi mathvariant="script">C</mml:mi>
</mml:math>
</inline-formula> is the Coulomb matrix expressed in this basis. A detailed derivation of <italic>M</italic> and <inline-formula id="inf15">
<mml:math id="m17">
<mml:mi mathvariant="script">C</mml:mi>
</mml:math>
</inline-formula> can be found in (<xref ref-type="bibr" rid="B17">Friedrich et&#x20;al., 2009</xref>). Note that while the sum <italic>n</italic>&#x2032;&#x2032; only stretches over occupied states, <italic>n</italic> and <italic>n</italic>&#x2032; cover all states. The square Coulomb matrix <inline-formula id="inf16">
<mml:math id="m18">
<mml:mi mathvariant="script">C</mml:mi>
</mml:math>
</inline-formula> with the dimensions of the size of the mixed-product basis is largely sparse, which allows for a significant reduction in the computational effort. The product <inline-formula id="inf17">
<mml:math id="m19">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mfenced open="&#x27e8;" close="&#x27e9;">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold">k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold">q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> is referred to &#x201c;Sparse matmul&#x201d; in <xref ref-type="fig" rid="F1">Figure&#x20;1</xref>; <xref ref-type="fig" rid="F3">Figures 3</xref>&#x2013;<xref ref-type="fig" rid="F5">5</xref>. The projector matrix <inline-formula id="inf18">
<mml:math id="m20">
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="&#x27e8;" close="&#x27e9;">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold">k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold">q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> has the dimensions of the size of the mixed-product basis and the number of states. The evaluation of this term is split into two components, one called &#x201c;inters. wave-prod&#x201d; and one called &#x201c;MT wave-prod&#x201d;, referring to the evaluation of this term either within the interstitial region or the muffin-tins. In order to apply <inline-formula id="inf19">
<mml:math id="m21">
<mml:msubsup>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mtext>exact</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> to the Hamiltonian in the LAPW basis it is transformed from the eigenspace to the LAPW basis by applying the overlap matrix of the LAPW matrix and the eigenvector matrix. In <xref ref-type="fig" rid="F1">Figure&#x20;1</xref>; <xref ref-type="fig" rid="F3">Figures 3</xref>&#x2013;<xref ref-type="fig" rid="F5">5</xref> the full evaluation of the non-local potential, i.e.,&#x20;the setup of the Coulomb matrix, the evaluation of <xref ref-type="disp-formula" rid="e2">Eq. 2</xref> and the transformation into the LAPW basis together, is denoted as &#x201c;non-local pot.&#x201d;</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Strong scaling behavior with OpenMP on a single AMD EPYC 7742&#x20;64 core processor. The overall FLEUR iteration is shown with brown pentagons, while the calculation of the nonlocal potential is shown in red triangles. The four remaining lines show the major parts of the nonlocal potential. The red triangles indicating the nonlocal potential largely coincide with the brown pentagons indicating the full runtime, making it difficult to see them. The parallel efficiency of slightly above 100<italic>%</italic> in the routine for the setup mixed-product basis for 4 and 8 nodes is explained by cache effects. Depending on the number of cores used the stride of the parallelized loops executed on each core is changed, which can reduce the number of cache misses if this stride coincides with certain array dimensions.</p>
</caption>
<graphic xlink:href="fmats-09-851458-g001.tif"/>
</fig>
<p>The numerical evaluation of <xref ref-type="disp-formula" rid="e2">Eq. 2</xref> represents the majority of the computational effort in a FLEUR calculation using hybrid functionals. In particular, the projection of the products of wave functions given in the LAPW basis set (<xref ref-type="disp-formula" rid="e1">Eq. 1</xref>) onto the mixed-product basis and the multiplication of these projections with the Coulomb matrix provide significant computational challenges. The implementation developed for this work relies on collecting data processed in the same way. It allows to exploit data parallelism on multiple levels, be it the use of a SIMD instruction set or an efficient and balanced use of multiple threads. Two significant changes have been made to the basic algorithm introduced in (<xref ref-type="bibr" rid="B7">Betzinger et&#x20;al., 2010</xref>). First, contrary to the previous implementation, the projection onto the mixed-product basis in the interstitial region is now calculated by Fourier transforming the wave-functions into real-space and multiplying pairs of wave-functions there, before transforming them back into <bold>G</bold>-space. By employing fast Fourier transformations, this reduces the complexity of this calculation from <inline-formula id="inf20">
<mml:math id="m22">
<mml:mi mathvariant="script">O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf21">
<mml:math id="m23">
<mml:mi mathvariant="script">O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, where <italic>n</italic>
<sub>
<bold>G</bold>
</sub> is the number of LAPW basis functions. Second, rather than calculating all elements of <inline-formula id="inf22">
<mml:math id="m24">
<mml:msubsup>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>exact</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> individually as vector-matrix-vector products of the Coulomb matrix and the mixed-product basis, the new implementation stacks groups of vectors of the mixed-product-basis into matrices and then calculates blocks of <inline-formula id="inf23">
<mml:math id="m25">
<mml:msubsup>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>exact</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> as a single matrix-matrix-matrix product. While these operations are mathematically identical, this new block-wise implementation is twice as fast as the element-wise implementation even on a single CPU core, which is due to its better utilization of the core&#x2019;s vector units. Additionally, the element-wise evaluation of this term experiences almost no speedup if multiple cores are used, while the speedup of the modern implementation is shown in <xref ref-type="fig" rid="F1">Figure&#x20;1</xref> in blue. Calculating the non-local potential on a single NVIDIA A100 GPU results in a speedup of 4&#xd7; compared to an AMD EPYC 7742 CPU for the NaCl system with 64&#x20;atoms.</p>
</sec>
<sec id="s2-2">
<title>2.2 Shared Memory Parallelization</title>
<p>To enable the utilization of supercomputers with complex memory hierarchies, we rely on two classes of parallelization. We use shared memory parallelization to make full use of many-core CPUs or GPUs. While distributed memory parallelization is employed to distribute the calculation over hundreds of compute nodes. Shared memory parallelization is realized by utilizing libraries for standard math problems such as matrix-multiplications or Fourier transformations whenever possible. Code parts that do not fall within the mold of any standard math problem were parallelized using OpenMP on CPUs and OpenACC on GPUs. In <xref ref-type="fig" rid="F1">Figure&#x20;1</xref> the strong scaling behavior on a single node is shown. For strong scaling a fixed-size problem is calculated with an increasing amount of resources and the resulting speedup is measured. Here, the speedup is defined as <inline-formula id="inf24">
<mml:math id="m26">
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2254;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>min</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:math>
</inline-formula>, which in the case of ideal scaling behaviour is <inline-formula id="inf25">
<mml:math id="m27">
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>ideal</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>min</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:math>
</inline-formula>, where <italic>T</italic>
<sub>
<italic>n</italic>
</sub> is the runtime with <italic>n</italic> cores, nodes or GPUs and <italic>n</italic>
<sub>min</sub> is the minimal value of <italic>n</italic> that was used in a calculation. This can be used to define the parallel efficiency as <inline-formula id="inf26">
<mml:math id="m28">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2254;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>ideal</mml:mtext>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:math>
</inline-formula>.</p>
<p>While some parts, such as the projection onto the mixed-product basis in the muffin-tins or the setup of the Coulomb matrix show excellent scaling, the speedup of the projection on the mixed-product basis in the interstitial exhibits a plateau around a speedup of 4 (see orange line). As mentioned previously, this algorithm relies on fast Fourier transformations, which have a low algorithmic intensity, meaning that only few calculations are performed compared to the number of load and store operations, making the algorithm more likely to be limited by the memory bandwidth rather than the available computational power, explaining the plateau in the speedup seen in <xref ref-type="fig" rid="F1">Figure&#x20;1</xref>. Up until 8 cores, the FFT still benefits from the additional compute resources, beyond that the FFTs are not limited anymore by the computational power, but rather by the memory bandwidth, which does not increase with the number of assigned&#x20;cores.</p>
</sec>
<sec id="s2-3">
<title>2.3 Distributed Memory Parallelization</title>
<p>The parallelism shown so far is limited to a single shared memory node and thus limited by the number of cores on a given node. Therefore, in order to scale the computational challenge posed by the hybrid exchange-correlation functionals to hundreds of nodes, we implemented three additional levels of distributed memory parallelism using MPI. The first two levels distribute the computations for different <bold>k</bold>- and <bold>q</bold>-points, while the third level parallelizes that of different occupied bands <italic>n</italic>&#x2032;&#x2032;. The distributed memory parallelization scheme is shown in <xref ref-type="fig" rid="F2">Figure&#x20;2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The distributed-memory parallelization of <xref ref-type="disp-formula" rid="e2">Eq. 2</xref> is divided into three levels. For each <bold>k</bold>-point the exact exchange is calculated as an independent problem. At a <bold>k</bold>-point <bold>k</bold>
<sub>
<italic>i</italic>
</sub>, the exact exchange is calculated as a sum over all <bold>q</bold>-points associated with <bold>k</bold>
<sub>
<italic>i</italic>
</sub>, building the <bold>kq</bold>-pairs. These first two levels require very little communication, i.e.,&#x20;copying the final results to their destination for the <bold>k</bold>-points and a reduction within the sub-communicator of each <bold>k</bold>-point for the <bold>q</bold>-point sum. The third level of distributed-memory parallelization calculates groups of occupied bands <italic>n</italic>&#x2032;&#x2032; in parallel. Here, a lot of inter-dependencies create a much larger communication demand compared to the first two levels. In a typical calculation of the non-local potential the workload is not uniformly distributed: Some <bold>k</bold>-points have more <bold>q</bold>-points than others and some <bold>kq</bold>-pairs might have more associated bands than others. The algorithm attempts to compensate this by assigning more nodes to heavier calculations.</p>
</caption>
<graphic xlink:href="fmats-09-851458-g002.tif"/>
</fig>
<p>The parallelization over <bold>k</bold>- and <bold>q</bold>-points requires very little communication and thus is very efficient, while the band-parallelization requires more communication. However, it allows us to limit the size of the largest matrix stored on a single node to <italic>n</italic>
<sub>basis size</sub> &#xd7; <italic>n</italic>
<sub>total bands</sub>, which then has a size on the order of <inline-formula id="inf27">
<mml:math id="m29">
<mml:mi mathvariant="script">O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atoms</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. This turns out to be the bottleneck that determines the largest system we can calculate on a given computing platform. With 90&#xa0;GB of memory per node we were able to calculate systems with up to 200&#x20;atoms.</p>
<p>
<xref ref-type="fig" rid="F3">Figure&#x20;3</xref> singles out the strong scaling behavior of this 3rd MPI level for a single <bold>k</bold>- and <bold>q</bold>-point. All code parts except for the setup of the Coulomb matrix show a good scaling behavior with a parallel efficiency of over 50% on either 64 CPU nodes or 64 GPUs. The scaling behavior of the Coulomb-matrix setup is not critical, since it does not dominate the run time of the calculation even on 256 nodes. Additionally, it only scales linearly with the number of <bold>k</bold>-points whereas the other parts of the nonlocal potential scale quadratically with the number of <bold>k</bold>-points. To investigate the performance of our algorithm with multiple <bold>k</bold>-points we show the strong scaling behavior for a NaCl supercell with 64 atoms, 10&#x20;<bold>k</bold>-points and 205&#x20;<bold>kq</bold>-pairs in <xref ref-type="fig" rid="F4">Figure&#x20;4</xref>. Here the scaling behavior of the Coulomb-matrix setup is the weakest once again, but it only accounts for less than 10<italic>%</italic>, even with 410 nodes. All other code parts show nearly perfect scaling. This is due to the fact that each <bold>kq</bold>-pair represents a largely independent problem that only requires little communication and the 3rd MPI level is only used with 205 and 410 nodes, since the parallelization over the 205&#x20;<bold>kq</bold>-pairs is preferred. SuperMUC-NG has two CPUs per node and therefore we assigned two MPI ranks to each node, resulting in a better performance compared to a one rank-per-node&#x20;setup.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Scaling behavior of systems with a single <bold>k</bold>-point on two types of architectures. Panels <bold>(A)</bold> and <bold>(B)</bold> show scaling of a 99-atom FeO supercell with a vacancy defect on the CPU-based SuperMUC-NG supercomputer, while <bold>(C)</bold> and <bold>(D)</bold> show the scaling behavior of a 120-atom GaAs supercell with an Al defect on JURECA&#x2019;s GPU partition. Panels <bold>(A)</bold> and <bold>(C)</bold> show the speedup, while <bold>(B)</bold> and <bold>(D)</bold> show the corresponding parallel efficiency.</p>
</caption>
<graphic xlink:href="fmats-09-851458-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Strong scaling behavior of multiple <bold>k</bold>-points for a 64 atom NaCl supercell with a potassium (K) defect. The system has 10&#x20;<bold>k</bold>-points and 205&#x20;<bold>kq</bold>-pairs. The super-scalar behavior is caused by the fact, that the 205&#x20;<bold>kq</bold>-pairs are not evenly distributed on 10 nodes (20 MPI). Some nodes are assigned more <bold>kq</bold>-pairs and therefore need longer while the others sit idle. This effect disappears for 205 or 410 nodes, which allow for a perfectly even and therefore more efficient distribution. For 41 nodes, the speedup and parallel efficiency of the Coulomb-matrix setup drop drastically. This is due to the fact, that the Coulomb-matrix setup does not have a <bold>q</bold>-dependence, while the number of nodes is chosen to be optimal for the evaluation of <xref ref-type="disp-formula" rid="e2">Eq. 2</xref>. For 10 nodes (20 MPI), all <bold>k</bold>-points can be calculated in parallel on 2 processes each, while for 41 nodes (82 MPI), it is only possible to calculate 2&#x20;<bold>k</bold>-points in parallel, so that each is distributed over 41 processes, leading to an inefficient parallelization. In practical calculations this is mitigated by including more nodes (e.g., 45), so that both <bold>k</bold>- and <bold>kq</bold>-parallelization are efficient. However, even with 41 nodes, the Coulomb-matrix setup only accounts for 6<italic>%</italic> of the total runtime.</p>
</caption>
<graphic xlink:href="fmats-09-851458-g004.tif"/>
</fig>
</sec>
<sec id="s2-4">
<title>2.4 Weak Scaling</title>
<p>While the meaning of strong scaling is very intuitive, it does not necessarily reflect real life applications. Being able to calculate a system with twenty atoms in a minute or less may not advance science significantly. Certainly, science is advanced by being able to calculate increasingly bigger, more inhomogeneous and more complex systems in a reasonable time frame. Weak scaling deals with the latter. As discussed in the introduction, the computational demand of a hybrid functional calculation scales with<disp-formula id="e3">
<mml:math id="m30">
<mml:mi mathvariant="script">O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atom</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>For simplicity and to focus on the ultimately limiting parallelization level, we use a single <bold>kq</bold>-pair and neglect the very efficient parallelization over different <bold>k</bold>- and <bold>q</bold>-vectors. In <xref ref-type="fig" rid="F5">Figure&#x20;5</xref> a gallium arsenide (GaAs) setup was scaled into supercells with a single nitrogen defect. Then, they were calculated with the parallelization chosen such that<disp-formula id="e4">
<mml:math id="m31">
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>nodes</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atoms</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atoms</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:math>
<label>(4)</label>
</disp-formula>where min(<italic>n</italic>
<sub>atoms</sub>) is the number of atoms in the smallest supercell.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Weak scaling behavior of FLEUR&#x2019;s hybrid functional calculations for a GaAs system which is scaled into a supercell with one Arsenic atom being substituted with Nitrogen. The <italic>y</italic>-axis shows the runtime for different code parts, while the bottom <italic>x</italic>-axis shows the number of nodes used. The top <italic>x</italic>-axis shows the number of atoms in that particular system.</p>
</caption>
<graphic xlink:href="fmats-09-851458-g005.tif"/>
</fig>
<p>With ideal weak scaling behavior the runtime should be constant regardless of the size of the unit cell, since the computational cost in <xref ref-type="disp-formula" rid="e3">Eq. 3</xref> is canceled out by the additional compute resources chosen in <xref ref-type="disp-formula" rid="e4">Eq. 4</xref>. <xref ref-type="fig" rid="F5">Figure&#x20;5</xref> shows that the hybrid functionals in FLEUR can be applied efficiently to a wide variety of system sizes. The time needed for the calculation of the nonlocal potential of the largest GaAs supercell is 9% larger compared to that of the smallest supercell and the full iteration runtime is 30% larger. The runtime does not monotonously increase as one would expect for the weak scaling of a simple algorithm performing a single task. In FLEUR, the situation is more complex, some parts of the code scale with <inline-formula id="inf28">
<mml:math id="m32">
<mml:mi mathvariant="script">O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atom</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, while others scale with <inline-formula id="inf29">
<mml:math id="m33">
<mml:mi mathvariant="script">O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atom</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>: While the setup of the mixed-product basis in the muffin-tin spheres grows with <inline-formula id="inf30">
<mml:math id="m34">
<mml:mi mathvariant="script">O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atom</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, its counterpart in the interstitial region grows with <inline-formula id="inf31">
<mml:math id="m35">
<mml:mi mathvariant="script">O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atom</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atom</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>. In the Coulomb-matrix setup, some parts such as the MT-MT interaction grow with <inline-formula id="inf32">
<mml:math id="m36">
<mml:mi mathvariant="script">O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atom</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, while e.g., &#x393;-point correction for in the interstitial grows with <inline-formula id="inf33">
<mml:math id="m37">
<mml:mi mathvariant="script">O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>atom</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>. For larger systems terms with a bigger scaling-exponent will be dominant, but in small systems the parts with the smaller scaling-exponents dominate the runtime. In these cases the choice of <xref ref-type="disp-formula" rid="e4">Eq. 4</xref> is not suitable, because the compute resources are increased faster than the computational complexity grows, leading to the initial dip in the overall runtime in <xref ref-type="fig" rid="F5">Figure&#x20;5</xref>.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Application to Rare-Earth Iron Garnets</title>
<p>Yttrium iron garnet (Y<sub>3</sub>Fe<sub>5</sub>O<sub>12</sub> or short <italic>YIG</italic>) is a complex ferrimagnetic insulator with a number of remarkable properties and applications, in the fields of magnonics (<xref ref-type="bibr" rid="B41">Serga et&#x20;al., 2010</xref>), ultra-low temperature physics (<xref ref-type="bibr" rid="B14">Demokritov et&#x20;al., 2006</xref>) and quantum computing (<xref ref-type="bibr" rid="B43">Tabuchi et&#x20;al., 2015</xref>). This success has sparked interest in a related class of materials, the so-called rare-earth-iron garnets (RIGs), where the yttrium atom in the YIG structure is replaced with an element of the lanthanide series. Here applications range from materials with giant magnetostriction (<xref ref-type="bibr" rid="B39">Sayetat, 1986</xref>) to spin Seebeck insulators (<xref ref-type="bibr" rid="B45">Uchida et&#x20;al., 2010</xref>). Despite great interest in these materials there is only a limited number of theoretical studies of their electronic structure. This is most likely due to the large unit cells with 160 atoms in the conventional and 80 atoms in the primitive unit&#x20;cell.</p>
<p>The typical unit cell of a garnet is shown in <xref ref-type="fig" rid="F6">Figure&#x20;6</xref>. The iron atoms in this structure have two types of environments. They are either in the centre of an octahedron or a tetrahedron spanned by neighbouring oxygen atoms. These different iron environments have a strong effect on the electronic structure, which is discussed in detail later in this paper. YIG and most RIGs are ferrimagnets, such that the magnetic moments of the 8 octahedral iron atoms point in the opposite direction with respect to the 12 tetrahedral iron atoms, which, for the RIGs discussed here, are aligned in parallel with the rare-earth elements. Only a very minor magnetic moment is induced in the yttrium and oxygen&#x20;atoms.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Unit cell of a garnet. Oxygen atoms are shown in red, while iron atoms are shown inside the grey polyhedra. The rare-earth or yttrium atoms are shown inside the golden dodecahedra. While the yttrium or rare-earth atoms are all symmetry equivalent the iron is present in two different environments. Structure from (<xref ref-type="bibr" rid="B49">Y3Fe5O12 crystal structure, 2012</xref>) and plotted with (<xref ref-type="bibr" rid="B32">Momma and Izumi, 2011</xref>).</p>
</caption>
<graphic xlink:href="fmats-09-851458-g006.tif"/>
</fig>
<sec id="s3-1">
<title>3.1 Electronic Structure</title>
<p>In order to understand how the choice of the exchange-correlation functional affects the electronic structure of YIG we calculated the density-of-states (DOS) with PBE and with PBE0. All calculations shown in the paper were performed on a 2 &#xd7; 2&#x20;&#xd7; 2&#x20;<bold>k</bold>-point grid. We confirmed that the DOS is converged with this grid by comparing the PBE results to results on a denser <bold>k</bold>-point grid. We use a smearing of <italic>&#x3c3;</italic> &#x3d; 0.136&#xa0;eV for all DOS calculations shown. The muffin-tin radii of Y, Gd, Tm, Fe and O are chosen to be <italic>r</italic>
<sub>Y/Gd/Tm</sub> &#x3d; 2.8&#x20;<italic>a</italic>
<sub>0</sub>, <italic>r</italic>
<sub>Fe</sub> &#x3d; 2.14&#x20;<italic>a</italic>
<sub>0</sub> and <italic>r</italic>
<sub>O</sub> &#x3d; 1.21&#x20;<italic>a</italic>
<sub>0</sub>. The structural information, e.g. the unit cell and the atomic positions used in this chapter are based on the experimental ones, exhibiting a <inline-formula id="inf34">
<mml:math id="m38">
<mml:mi>I</mml:mi>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>d</mml:mi>
</mml:math>
</inline-formula>-structure (<xref ref-type="bibr" rid="B49">Y3Fe5O12 crystal structure, 2012</xref>; <xref ref-type="bibr" rid="B19">Gd3FeO12 crystal structure, 2012</xref>; <xref ref-type="bibr" rid="B44">Tm3Fe5O12 crystal structure, 2012</xref>).</p>
<p>As expected, with a value of 0.44&#xa0;eV, PBE massively underestimates the experimental band gap of 2.8&#xa0;eV (<xref ref-type="bibr" rid="B25">Larsen and Metselaar, 1975</xref>), while PBE0 predicts an improved band gap of 1.83&#xa0;eV. However, the experimental value relies on optical measurements, which are not sensitive to all transitions, potentially missing certain states and thus overestimating the real band&#x20;gap.</p>
<p>In <xref ref-type="fig" rid="F7">Figure&#x20;7A</xref>) the DOS of YIG is calculated using PBE as an exchange-correlation functional. In this figure, the antiferrimagnetic alignment of the iron atoms is visible: the occupied states associated with the tetrahedral iron atoms are mainly in the spin-up channel and the unoccupied ones are in the spin-down channel, while for the octahedral iron atoms the situation is reversed: below the Fermi level the octahedral iron states are mostly in the spin-down channel and above it in the spin-up one. Most states associated with the oxygen atoms are occupied, while the yttrium states are largely unoccupied. Below the Fermi level, the DOS in the interstitial region closely follows the oxygen DOS. Additionally, the DOS associated with both iron types also coincide with the oxygen and interstitial DOS. This indicates that the 2<italic>p</italic>-states of the oxygen and the 3<italic>d</italic>-states of iron hybridize for both iron environments. This analysis is supported by the number of valence electrons found in the different muffin-tin spheres, which are 6.5 and 6.2 electrons for iron atoms in the tetrahedral and octhedral environments, respectively, 1.1 valence electrons in the sphere of yttrium, and an average of 3.7 valence electrons in the spheres of oxygen. The large number of 164.1 electrons in the interstitial region additionally indicates a high degree of de-localization of these states. For the unoccupied octahedral iron states in contrast, we can see a clear signature of simple crystal field splitting of localized <italic>d</italic>-states: the three <italic>t</italic>
<sub>2<italic>g</italic>
</sub>-states shift down and the two e<sub>g</sub>-states shift up leading to two distinct peaks with the lower one containing three and the higher one containing two states. Similarly, for the unoccupied tetrahedral Fe <italic>d</italic>-levels the <italic>e</italic>-states are shifted down, while the <italic>t</italic>
<sub>2</sub>-states are shifted up. This separation however, is not so clear as the shifts are smaller, the peaks still overlap and another splitting due to next-nearest neighbors can be&#x20;seen.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>DOS of YIG calculated with PBE in <bold>(A)</bold> and with PBE0 in <bold>(B)</bold> on a 2 &#xd7; 2 &#xd7; 2&#x20;<bold>k</bold>-grid. In both calculations we use a <inline-formula id="inf35">
<mml:math id="m39">
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>max</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>4.5</mml:mn>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. The mixed-product basis for the calculation in <bold>(B)</bold> uses a eigenvalue threshold of <italic>&#x3bd;</italic> &#x3d; 10<sup>&#x2212;4</sup> and an l<sub>MPB</sub> &#x3d; 16 cutoff for the spherical harmonics.</p>
</caption>
<graphic xlink:href="fmats-09-851458-g007.tif"/>
</fig>
<p>In <xref ref-type="fig" rid="F7">Figure&#x20;7B</xref>) we show the DOS calculated using the hybrid exchange-correlation functional PBE0. The results are qualitatively different from the PBE results with the most significant change seen in the different behavior of the two types of Fe in the PBE results: While the occupied tetrahedral iron 3<italic>d</italic>-states still hybridize with the 2<italic>p</italic>-states of the surrounding oxygen atoms, most of the octahedral iron 3<italic>d</italic>-states are now strongly localized and form a double peak in the DOS at around &#x2212;6.5&#xa0;eV.</p>
<p>Such a localization effect of the <italic>d</italic>-states can also be reproduced in a PBE&#x2b;U treatment (<xref ref-type="bibr" rid="B12">Chen et&#x20;al., 2021</xref>). However, in these simulations the <italic>d</italic>-states of both Fe types show the same behavior. The different tendency to localize can also be seen in these simulation by the different values of <italic>U</italic> used for the different atoms to achieve the localization. Hence, the result strongly suggests that the local Coulomb interaction exhibits different strength in the two environments of Fe. This effect can be caused by the different initial localization of the <italic>d</italic>-states as well as by different interactions and screening effects of the surrounding. Such a difference can also be seen in the unoccupied spectrum which is again dominated by a crystal field splitting of the <italic>d</italic>-states. However, in the octahedral environment this effect is again much clearer while the tetrahedral <italic>d</italic>-states form a broad band with several peaks also indicating next-nearest neighbor effects.</p>
<p>Finally, we would like to point out that the octahedral Fe <italic>d</italic>-states show a rather complex sub-structure with a large peak at &#x2212; 7&#xa0;eV and a minor peak at &#x2212; 8&#xa0;eV. This is not a crystal field splitting but rather shows that different states with a different degree of localization are formed. While the lower peak is clearly separated from the O <italic>p</italic>-states, some remaining hybridization can be identified for the larger&#x20;peak.</p>
<p>Further investigations of the consequences of these differences between the Fe atoms is beyond the scope of this paper, but we expect this electronic structure to have some influence, e.g. on magnetic interactions and the transition temperature.</p>
</sec>
<sec id="s3-2">
<title>3.2 Magnetic Moment</title>
<p>In the introduction we discussed that some key applications of YIG are related to its magnetic properties. Therefore, we want to investigate the precision of our predictions for magnetic properties with different exchange-correlation functionals. In <xref ref-type="table" rid="T1">Table&#x20;1</xref> we compare the magnetic moments predicted for the different iron atom types. We use the magnetic moment inside the muffin-tin sphere to assign the moment to a specific atom. Therefore, the magnetic moment depends on the choice of the radius of the muffin-tin sphere and, strictly speaking, is not uniquely defined. The magnetization calculated for the oxygen and yttrium atoms is negligible regardless of the computational method used. The total magnetic moment per unit formula was 5&#xa0;&#x3bc;<sub>B</sub> for every functional. This agreement is expected, since YIG is a magnetic insulator, which constrains the total magnetization per unit cell to integer values.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The magnetic moments within the different muffin-tins of both Fe types in units of <italic>&#x3bc;</italic>
<sub>B</sub>.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="2" align="left"/>
<th align="center">Fe tetra. [<italic>&#x3bc;</italic>
<sub>B</sub>]</th>
<th align="center">Fe octa. [<italic>&#x3bc;</italic>
<sub>B</sub>]</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="2" align="left">PBE</td>
<td align="char" char=".">3.52</td>
<td align="char" char=".">&#x2212;3.64</td>
</tr>
<tr>
<td colspan="2" align="left">PBE0</td>
<td align="char" char=".">3.83</td>
<td align="char" char=".">&#x2212;4.01</td>
</tr>
<tr>
<td align="left">PBE &#x2b; U</td>
<td align="left">
<xref ref-type="bibr" rid="B34">Nakamoto et&#x20;al. (2017)</xref>
</td>
<td align="char" char=".">4.10</td>
<td align="char" char=".">&#x2212;4.20</td>
</tr>
<tr>
<td align="left">QSGW</td>
<td align="left">
<xref ref-type="bibr" rid="B3">Barker et&#x20;al. (2020)</xref>
</td>
<td align="char" char=".">3.93</td>
<td align="char" char=".">&#x2212;4.17</td>
</tr>
<tr>
<td align="left">exp. <inline-formula id="inf36">
<mml:math id="m40">
<mml:mspace width="0.17em"/>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<xref ref-type="bibr" rid="B37">Rodic et&#x20;al. (1999)</xref>
</td>
<td align="char" char=".">3.95</td>
<td align="char" char=".">&#x2212;4.01</td>
</tr>
<tr>
<td align="left">exp. <inline-formula id="inf37">
<mml:math id="m41">
<mml:mspace width="0.17em"/>
<mml:mi>I</mml:mi>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>d</mml:mi>
</mml:math>
</inline-formula>
</td>
<td align="left">
<xref ref-type="bibr" rid="B37">Rodic et&#x20;al. (1999)</xref>
</td>
<td align="char" char=".">5.37</td>
<td align="char" char=".">&#x2212;4.11</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>While PBE predicts the magnetic moments of the two iron types within only &#xb1;0.5&#xa0;&#x3bc;<sub>B</sub> of the experimental value for the <inline-formula id="inf38">
<mml:math id="m42">
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> crystal structure, the predictions by PBE0 are even closer to the experimental those results. This again can be understood by the observed tendency to localize the Fe <italic>d</italic>-states and compares very well to magnetic moments predicted by Barker et&#x20;al. (<xref ref-type="bibr" rid="B3">Barker et&#x20;al., 2020</xref>) obtained using QSGW, another highly precise electronic structure method. The slight difference in the magnetic moments between the QSGW and PBE0 approach we account to the different choice of the muffin-tin radii and the different degree of localization of the Fe <italic>d</italic>-states. Interestingly, in the comparison of PBE0 to QSGW in the case of this octahedral iron, the magnetic moment for this octahedral Fe agrees better with the experimental results in the <inline-formula id="inf39">
<mml:math id="m43">
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> crystal structure than the QSGW value, supporting our findings of a slightly stronger localized <italic>d</italic>-wave function in the case of the PBE0 functional. We note that all theoretical results were calculated for the cubic <inline-formula id="inf40">
<mml:math id="m44">
<mml:mi>I</mml:mi>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>d</mml:mi>
</mml:math>
</inline-formula> crystal structure and the magnetic moments of Fe in the tetrahedral environment agree quite well with each other but also with the experimental values of Fe in the trigonal <inline-formula id="inf41">
<mml:math id="m45">
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> structure. On the other hand, the experimental Fe moment in the <inline-formula id="inf42">
<mml:math id="m46">
<mml:mi>I</mml:mi>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>d</mml:mi>
</mml:math>
</inline-formula> symmetry is completely off. It shows a moment of 5.37&#xa0;&#x3bc;<sub>B</sub>. This value seems unrealistic since it is higher than that of a free Fe<sup>3&#x2b;</sup> ion (&#x223c; 5&#xa0;&#x3bc;<sub>B</sub>), while the presence of hybridization with oxygen is expected to lower the moment further. We conclude that further experimental efforts are needed to analyze the structure-magnetism relationship of&#x20;YIG.</p>
</sec>
<sec id="s3-3">
<title>3.3 Rare-earth-iron Garnets</title>
<p>As two representatives of the rare-earth-iron garnet group, we chose to examine Gd<sub>3</sub>Fe<sub>5</sub>O<sub>12</sub> (GdIG) and Tm<sub>3</sub>Fe<sub>5</sub>O<sub>12</sub> (TmIG) more closely. We selected these materials, because a lot of interesting experimental (<xref ref-type="bibr" rid="B15">Fechine et&#x20;al., 2008</xref>; <xref ref-type="bibr" rid="B36">Phan et&#x20;al., 2009</xref>; <xref ref-type="bibr" rid="B26">Lassri et&#x20;al., 2011</xref>; <xref ref-type="bibr" rid="B27">Lee et&#x20;al., 2020</xref>; <xref ref-type="bibr" rid="B46">Vilela et&#x20;al., 2020</xref>; <xref ref-type="bibr" rid="B47">Vu et&#x20;al., 2020</xref>) and even some theoretical work using the FLAPW method (<xref ref-type="bibr" rid="B26">Lassri et&#x20;al., 2011</xref>) has been published for these materials.</p>
<p>In <xref ref-type="fig" rid="F8">Figure&#x20;8</xref> we present the density of states for GdIG and TmIG calculated with the PBE0&#x20;exchange-correlation functional. Reaching numerical self-consistency for TmIG was challenging with PBE, which is the starting point for any PBE0 calculation. We achieved self-consistency by using a few hundred straight mixing iterations with a low mixing parameter, followed by a set of Anderson mixing iterations until convergence was reached. With a converged PBE as a starting density the convergence of PBE0 is straight forward. This difficult convergence is caused by the metallic behavior of TmIG with PBE as a functional. After the non-local potential is included, a gap opens up and all later density convergence cycles do not exhibit this problematic metallic behavior. GdIG converged without problems both for PBE and&#x20;PBE0.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>The density-of-states is calculated for GdIG in <bold>(A)</bold> and for TmIG in <bold>(B)</bold> using PBE0 on a 2 &#xd7; 2 &#xd7; 2&#x20;<bold>k</bold>-point grid. Both calculations were performed with a <inline-formula id="inf43">
<mml:math id="m47">
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>max</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>4.5</mml:mn>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> and the mixed-product basis was setup using <italic>&#x3bd;</italic> &#x3d; 10<sup>&#x2212;4</sup> and <italic>l</italic>
<sub>MPB</sub> &#x3d; 16. Both band gaps are 1.7&#xa0;eV and marked in red. The Gd states are fully occupied for the majority spin-channel and fully unoccupied for the minority spin-channel. The Tm spin-up channel is also fully occupied while the minority spin channel is only partially occupied.</p>
</caption>
<graphic xlink:href="fmats-09-851458-g008.tif"/>
</fig>
<p>For GdIG the band gap was calculated to be 1.7&#xa0;eV with PBE0. Literature values obtained using PBE &#x2b; U suggest a gap of 1.6&#xa0;eV (<xref ref-type="bibr" rid="B34">Nakamoto et&#x20;al., 2017</xref>). For TmIG we also predict a band gap of 1.7&#xa0;eV using PBE0. To our knowledge, this is the first prediction for the band gap of TmIG. We are not aware of any experimental results regarding the band gap in either system.</p>
<p>The electronic structure of these two systems has a few striking similarities with that of YIG. The 3<italic>d</italic>-states of both types of iron atoms hybridize with the oxygen 2<italic>p</italic>-states in PBE, while with PBE0 the octahedral iron states show localization and a strong shift to lower energies. This again highlights the difference of the tetrahedral and octahedral oxygen environment of the iron atoms causing different effective interactions at these atoms and casting doubt on simple PBE&#x2b;U predictions for these garnet systems. For the unoccupied octahedral iron states we can see the typical signature of crystal field splitting and in the tetrahedral case this signature is weaker. The additional 4<italic>f</italic>-states of the rare-earth elements in the spin-up channel are strongly localized in PBE. In PBE0 they show a slightly larger bandwidth, indicating increased hybridization with the oxygen 2<italic>p</italic>-states which could be understood due to the decrease of hybridization of these states with the octahedral Fe <italic>d</italic>-states. As expected, Gd has no occupied 4<italic>f</italic>-states in the spin-down channel, while the 4<italic>f</italic>-states of Tm are partially occupied, causing a metallic behavior in PBE. In PBE0 the increased interaction provided by the exchange term opens a gap in the Tm 4<italic>f</italic>-band.</p>
<p>In <xref ref-type="table" rid="T2">Table&#x20;2</xref> the magnetic moments of all atom types are given. For GdIG we predict a total magnetization per formula unit of 16.0<italic>&#xa0;</italic>&#x3bc;<sub>B</sub> and for TmIG we predict 1.75&#xa0;&#x3bc;<sub>B</sub> for PBE as well as PBE0. Notice, that the formula unit contains 20 atoms, while the primitive unit cell contains 80. This means, while the magnetic moment per formula unit is not integer, it is integer per unit&#x20;cell.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Predicted magnetic moments of GdIG and TmIG for each atom type, given in units of [<italic>&#x3bc;</italic>
<sub>B</sub>].</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="2" align="left"/>
<th align="center">Fe tetra [<italic>&#x3bc;</italic>
<sub>B</sub>]</th>
<th align="center">Fe octa [<italic>&#x3bc;</italic>
<sub>B</sub>]</th>
<th align="center">Gd/Tm [<italic>&#x3bc;</italic>
<sub>B</sub>]</th>
<th align="center">O [<italic>&#x3bc;</italic>
<sub>B</sub>]</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">Gd<sub>3</sub>Fe<sub>5</sub>O<sub>12</sub>
</td>
<td align="left">PBE</td>
<td align="char" char=".">&#x2212;3.54</td>
<td align="char" char=".">3.69</td>
<td align="char" char=".">6.88</td>
<td align="char" char=".">&#x2212;0.06</td>
</tr>
<tr>
<td align="left">PBE0</td>
<td align="char" char=".">&#x2212;3.85</td>
<td align="char" char=".">4.04</td>
<td align="char" char=".">6.94</td>
<td align="char" char=".">&#x2212;0.06</td>
</tr>
<tr>
<td align="left"/>
<td align="left">PBE &#x2b; U (<xref ref-type="bibr" rid="B34">Nakamoto et&#x20;al., 2017</xref>)</td>
<td align="char" char=".">&#x2212;4.1</td>
<td align="char" char=".">4.2</td>
<td align="char" char=".">7.0</td>
<td align="center">&#x2212;</td>
</tr>
<tr>
<td rowspan="3" align="left">Tm<sub>3</sub>Fe<sub>5</sub>O<sub>12</sub>
</td>
<td align="left">PBE</td>
<td align="char" char=".">&#x2212;3.298</td>
<td align="char" char=".">3.59</td>
<td align="char" char=".">1.61</td>
<td align="char" char=".">&#x2212;0.04</td>
</tr>
<tr>
<td align="left">PBE0</td>
<td align="char" char=".">&#x2212;3.82</td>
<td align="char" char=".">4.01</td>
<td align="char" char=".">1.93</td>
<td align="char" char=".">&#x2212;0.05</td>
</tr>
<tr>
<td align="left">PBE &#x2b; U (<xref ref-type="bibr" rid="B34">Nakamoto et&#x20;al., 2017</xref>)</td>
<td align="char" char=".">&#x2212;4.1</td>
<td align="char" char=".">4.2</td>
<td align="char" char=".">1.9</td>
<td align="center">&#x2212;</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The predicted total magnetic moments are in exact agreement with experimental results for GdIG (<xref ref-type="bibr" rid="B20">Geller et&#x20;al., 1965</xref>), while they are in good agreement with the experimental value of 1.2&#xa0;&#x3bc;<sub>B</sub> for the TmIG. This experimental value would correspond to a total magnetic moment of 4.8&#xa0;&#x3bc;<sub>B</sub> for the primitive unit cell. PBE &#x2b; U shows a tendency to predict larger magnetic moments for almost all atoms: 4.2&#xa0;&#x3bc;<sub>B</sub> for the octahedral iron, &#x2212; 4.1&#xa0;&#x3bc;<sub>B</sub> for the tetrahedral iron, 7.0&#xa0;&#x3bc;<sub>B</sub> for Gd and 1.9&#xa0;&#x3bc;<sub>B</sub> for Tm (<xref ref-type="bibr" rid="B34">Nakamoto et&#x20;al., 2017</xref>).</p>
</sec>
</sec>
<sec id="s4">
<title>4 Conclusion</title>
<p>In this article we presented a highly scalable implementation of hybrid exchange-correlation functionals in the LAPW basis. In this work we focused on the scalable implementation of the Hartree-Fock exact exchange, which corresponds to the implementation of the PBE0 functional, but screened functionals like HSE06 are related by an additional fast computation of a smooth function. The combination of shared and distributed memory parallelization allows to calculate a broad range of systems with high efficiency. Combining all three MPI levels gives us an outlook on the scaling potential of this algorithm. If we were for example to calculate the GaAs system with 120 atoms and we would use 8&#x20;<bold>k</bold>-points we would get 125&#x20;<bold>kq</bold>-pairs. <xref ref-type="fig" rid="F3">Figure&#x20;3</xref> shows that for this system a single <bold>kq</bold>-pair has a good parallel performance even if distributed over more than 32 GPUs. Therefore, it is reasonable to assume that the calculation of the nonlocal potential for a system with 8&#x20;<bold>k</bold>-points would still have good scaling with <inline-formula id="inf44">
<mml:math id="m48">
<mml:mn>32</mml:mn>
<mml:mfrac>
<mml:mrow>
<mml:mtext>GPUs</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">k</mml:mi>
<mml:mi mathvariant="bold">q</mml:mi>
<mml:mo>-</mml:mo>
<mml:mtext>pair</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>125</mml:mn>
<mml:mtext>&#x2009;GPUs</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>4000</mml:mn>
</mml:math>
</inline-formula> GPUs, which is &#x223c; 250 more than the 44 PetaFLOP JUWELS Booster Module has to offer. This not only allows the code to run on the supercomputers currently available, it also gives us confidence that our code can make good practical use of future exascale machines. Here, making good practical use of a supercomputer does not necessarily mean sending jobs which queue for weeks-on-end and then scale to every single core which the machine has to offer, but rather that we can efficiently use significant portions of the machine to investigate interesting and meaningful systems.</p>
<p>Using the new implementation of the hybrid functional code, we performed simulations of the electronic structure of iron based garnet materials. The significant improvement in the obtained band gap as well as the changes in the electronic structures discussed in detail demonstrate the significance and power of this treatment for these technological relevant material class. Our results suggests an experimental reevaluation of the structure-magnetism relation of the yttrium iron garnet (YIG), Y<sub>3</sub>Fe<sub>5</sub>O<sub>12</sub>.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>MR, DW, and GM performed and analyzed the initial performance measurements. MR, DW, and GM designed the 3-Level MPI parallelism with feedback form CT and MM. CT and MM suggested solutions for the performance bottlenecks in MPIs one-sided communication. MR, DW, GM implemented shared-memory parallelism using OpenMP, while CT contributed memory blocking for certain OpenMP kernels. MR performed the performance measurements and the electronic structure calculation for the garnet materials. JB and SB helped with the analysis and understanding of the electronic structure of the garnet materials. JB created <xref ref-type="fig" rid="F5">Figure&#x20;5</xref>.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>The authors gratefully acknowledge the computing time granted through JARA on the supercomputer JURECA at Forschungszentrum J&#xfc;lich as well as the Gauss Centre for Supercomputing e.V. (<ext-link ext-link-type="uri" xlink:href="http://www.gauss-centre.eu">www.gauss-centre.eu</ext-link>). for funding this project by providing computing time on the SuperMUC-NG Supercomputer. The work is funded by the JARA-CSD School for Simulation and Data Science (SSD) and by the MaX Center of Excellence funded by the EU through the H2020-INFRAEDI-2018 (project: GA 824143).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>We would like to acknowledge fruitful discussions with Christoph Friedrich.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alberi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Nardelli</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Zakutayev</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mitas</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Curtarolo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jain</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>The 2019 Materials by Design Roadmap</article-title>. <source>J.&#x20;Phys. D: Appl. Phys.</source> <volume>52</volume>, <fpage>013001</fpage>. <pub-id pub-id-type="doi">10.1088/1361-6463/aad926</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andersen</surname>
<given-names>O. K.</given-names>
</name>
<name>
<surname>Woolley</surname>
<given-names>R. G.</given-names>
</name>
</person-group> (<year>1973</year>). <article-title>Muffin-Tin Orbitals and Molecular Calculations: General Formalism</article-title>. <source>Mol. Phys.</source> <volume>26</volume>, <fpage>905</fpage>&#x2013;<lpage>927</lpage>. <pub-id pub-id-type="doi">10.1080/00268977300102171</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barker</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pashov</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Jackson</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Electronic Structure and Finite Temperature Magnetism of Yttrium Iron Garnet</article-title>. <source>Electron. Struct.</source> <volume>2</volume>, <fpage>044002</fpage>. <pub-id pub-id-type="doi">10.1088/2516-1075/abd097</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barnes</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Kurth</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Carrier</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wichmann</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Prendergast</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kent</surname>
<given-names>P. R. C.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Improved Treatment of Exact Exchange in Quantum Espresso</article-title>. <source>Computer Phys. Commun.</source> <volume>214</volume>, <fpage>52</fpage>&#x2013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.1016/j.cpc.2017.01.008</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Becke</surname>
<given-names>A. D.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>A New Mixing of Hartree-Fock and Local Density&#x2010;Functional Theories</article-title>. <source>J.&#x20;Chem. Phys.</source> <volume>98</volume>, <fpage>1372</fpage>&#x2013;<lpage>1377</lpage>. <pub-id pub-id-type="doi">10.1063/1.464304</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Becke</surname>
<given-names>A. D.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Perspective: Fifty Years of Density-Functional Theory in Chemical Physics</article-title>. <source>J.&#x20;Chem. Phys.</source> <volume>140</volume>, <fpage>18A301</fpage>. <pub-id pub-id-type="doi">10.1063/1.4869598</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Betzinger</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Friedrich</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bl&#xfc;gel</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Hybrid Functionals within the All-Electron FLAPW Method: Implementation and Applications of PBE0</article-title>. <source>Phys. Rev. B</source> <volume>81</volume>, <fpage>195117</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevB.81.195117</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blaha</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Schwarz</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Tran</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Laskowski</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Madsen</surname>
<given-names>G. K. H.</given-names>
</name>
<name>
<surname>Marks</surname>
<given-names>L. D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Wien2k: An APW&#x2b;lo Program for Calculating the Properties of Solids</article-title>. <source>J.&#x20;Chem. Phys.</source> <volume>152</volume>, <fpage>074101</fpage>. <pub-id pub-id-type="doi">10.1063/1.5143061</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boys</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Shavitt</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>1959</year>). <article-title>Technical Report WIS-AF-13</article-title>, <publisher-name>University of Wisconsin</publisher-name>. </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Burke</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Perspective on Density Functional Theory</article-title>. <source>J.&#x20;Chem. Phys.</source> <volume>136</volume>, <fpage>150901</fpage>. <pub-id pub-id-type="doi">10.1063/1.4704546</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carnimeo</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Baroni</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Giannozzi</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Fast Hybrid Density-Functional Computations Using Plane-Wave Basis Sets</article-title>. <source>Electron. Struct.</source> <volume>1</volume>, <fpage>015009</fpage>. <pub-id pub-id-type="doi">10.1088/2516-1075/aaf7d4</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>H.-S.</given-names>
</name>
<name>
<surname>Lo</surname>
<given-names>F.-Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Spin-Dependent Optical Transitions in Yttrium Iron Garnet</article-title>. <source>Mater. Res. Express</source> <volume>8</volume>, <fpage>026101</fpage>. <pub-id pub-id-type="doi">10.1088/2053-1591/abe013</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cramer</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Truhlar</surname>
<given-names>D. G.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Density Functional Theory for Transition Metals and Transition Metal Chemistry</article-title>. <source>Phys. Chem. Chem. Phys.</source> <volume>11</volume>, <fpage>10757</fpage>&#x2013;<lpage>10816</lpage>. <pub-id pub-id-type="doi">10.1039/B907148B</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Demokritov</surname>
<given-names>S. O.</given-names>
</name>
<name>
<surname>Demidov</surname>
<given-names>V. E.</given-names>
</name>
<name>
<surname>Dzyapko</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Melkov</surname>
<given-names>G. A.</given-names>
</name>
<name>
<surname>Serga</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Hillebrands</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>Bose-Einstein Condensation of Quasi-Equilibrium Magnons at Room Temperature under Pumping</article-title>. <source>Nature</source> <volume>443</volume>, <fpage>430</fpage>&#x2013;<lpage>433</lpage>. <pub-id pub-id-type="doi">10.1038/nature05117</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fechine</surname>
<given-names>P. B. A.</given-names>
</name>
<name>
<surname>Moretzsohn</surname>
<given-names>R. S. T.</given-names>
</name>
<name>
<surname>Costa</surname>
<given-names>R. C. S.</given-names>
</name>
<name>
<surname>Derov</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>J.&#x20;W.</given-names>
</name>
<name>
<surname>Drehman</surname>
<given-names>A. J.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Magnego-Dielectric Properties of the Y3Fe5O12 and Gd3Fe5O12 Dielectric Ferrite Resonator Antennas</article-title>. <source>Microw. Opt. Technol. Lett.</source> <volume>50</volume>, <fpage>2852</fpage>&#x2013;<lpage>2857</lpage>. <pub-id pub-id-type="doi">10.1002/mop.23824</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="web">
<collab>Fleur</collab> (<year>2021</year>). <article-title>Forschungszentrum J&#xfc;lich - PGI-1 &#x26; IAS-1</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.flapw.de">https://www.flapw.de</ext-link>
</comment>. </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friedrich</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Schindlmayr</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bl&#xfc;gel</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Efficient Calculation of the Coulomb Matrix and its Expansion Around k&#x3d;0 within the FLAPW Method</article-title>. <source>Computer Phys. Commun.</source> <volume>180</volume>, <fpage>347</fpage>&#x2013;<lpage>359</lpage>. <pub-id pub-id-type="doi">10.1016/j.cpc.2008.10.009</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garza</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Scuseria</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Predicting Band Gaps with Hybrid Density Functionals</article-title>. <source>J.&#x20;Phys. Chem. Lett.</source> <volume>7</volume>, <fpage>4165</fpage>&#x2013;<lpage>4170</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jpclett.6b01807</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<collab>Gd3FeO12 crystal structure</collab> (<year>2012</year>). <source>Datasheet from &#x201c;Pauling File Multinaries Edition &#x2013; 2012&#x201d; in Springermaterials</source> (<publisher-loc>Japan</publisher-loc>: <publisher-name>Springer-Verlag Berlin Heidelberg &#x26; Material Phases Data System (MPDS), Switzerland &#x26; National Institute for Materials Science NIMS</publisher-name>). <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://materials.springer.com/isp/crystallographic/docs/sd_0308674">https://materials.springer.com/isp/crystallographic/docs/sd_0308674</ext-link>. Copyright 2016</comment> . </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Geller</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Remeika</surname>
<given-names>J.&#x20;P.</given-names>
</name>
<name>
<surname>Sherwood</surname>
<given-names>R. C.</given-names>
</name>
<name>
<surname>Williams</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Espinosa</surname>
<given-names>G. P.</given-names>
</name>
</person-group> (<year>1965</year>). <article-title>Magnetic Study of the Heavier Rare-Earth Iron Garnets</article-title>. <source>Phys. Rev.</source> <volume>137</volume>, <fpage>A1034</fpage>&#x2013;<lpage>A1038</lpage>. <pub-id pub-id-type="doi">10.1103/PhysRev.137.A1034</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guidon</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schiffmann</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Hutter</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>VandeVondele</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Ab Initio Molecular Dynamics Using Hybrid Density Functionals</article-title>. <source>J.&#x20;Chem. Phys.</source> <volume>128</volume>, <fpage>214104</fpage>. <pub-id pub-id-type="doi">10.1063/1.2931945</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hakala</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Foster</surname>
<given-names>A. S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Computationally Efficient Implementation of Hybrid Functionals in SIESTA</article-title>. <source>Tech. Rep.</source> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ihrig</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Wieferink</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>I. Y.</given-names>
</name>
<name>
<surname>Ropo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Rinke</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Accurate Localized Resolution of Identity Approach for Linear-Scaling Hybrid Density Functionals and for Many-Body Perturbation Theory</article-title>. <source>New J.&#x20;Phys.</source> <volume>17</volume>, <fpage>093020</fpage>. <pub-id pub-id-type="doi">10.1088/1367-2630/17/9/093020</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krukau</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Vydrov</surname>
<given-names>O. A.</given-names>
</name>
<name>
<surname>Izmaylov</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Scuseria</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Influence of the Exchange Screening Parameter on the Performance of Screened Hybrid Functionals</article-title>. <source>J.&#x20;Chem. Phys.</source> <volume>125</volume>, <fpage>224106</fpage>. <pub-id pub-id-type="doi">10.1063/1.2404663</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Larsen</surname>
<given-names>P. K.</given-names>
</name>
<name>
<surname>Metselaar</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1975</year>). <article-title>Defects and the Electronic Properties of Y3Fe5O12</article-title>. <source>J.&#x20;Solid State. Chem.</source> <volume>12</volume>, <fpage>253</fpage>&#x2013;<lpage>258</lpage>. <pub-id pub-id-type="doi">10.1016/0022-4596(75)90315-1</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lassri</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hlil</surname>
<given-names>E. K.</given-names>
</name>
<name>
<surname>Prasad</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Krishnan</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Magnetic and Electronic Properties of Nanocrystalline Gd3Fe5O12 Garnet</article-title>. <source>J.&#x20;Solid State. Chem.</source> <volume>184</volume>, <fpage>3216</fpage>&#x2013;<lpage>3220</lpage>. <pub-id pub-id-type="doi">10.1016/j.jssc.2011.09.034</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Flores</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bagu&#xe9;s</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Probing the Source of the Interfacial Dzyaloshinskii-Moriya Interaction Responsible for the Topological Hall Effect in Metal/Tm3Fe5O12 Systems</article-title>. <source>Phys. Rev. Lett.</source> <volume>124</volume>, <fpage>107201</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevLett.124.107201</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lejaeghere</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Bihlmayer</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bj&#xf6;rkman</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Blaha</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bl&#xfc;gel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Blum</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Reproducibility in Density Functional Theory Calculations of Solids</article-title>. <source>Science</source> <volume>351</volume>, <fpage>aad3000</fpage>. <pub-id pub-id-type="doi">10.1126/science.aad3000</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Levchenko</surname>
<given-names>S. V.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wieferink</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Johanni</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rinke</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Blum</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Hybrid Functionals for Large Periodic Systems in an All-Electron, Numeric Atom-Centered Basis Framework</article-title>. <source>Computer Phys. Commun.</source> <volume>192</volume>, <fpage>60</fpage>&#x2013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1016/j.cpc.2015.02.021</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Efficient Hybrid Density Functional Calculations for Large Periodic Systems Using Numerical Atomic Orbitals</article-title>. <source>J.&#x20;Chem. Theor. Comput.</source> <volume>17</volume>, <fpage>222</fpage>&#x2013;<lpage>239</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jctc.0c00960</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mandal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kar</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kloeffel</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Meyer</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Nair</surname>
<given-names>N. N.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Improving the Scaling and Performance of Multiple Time Stepping Based Molecular Dynamics with Hybrid Density Functionals</article-title>. <source>ArXiv</source>, <fpage>07670</fpage>. <pub-id pub-id-type="doi">10.1002/jcc.26816</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Momma</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Izumi</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>VESTA 3 for Three-Dimensional Visualization of Crystal, Volumetric and Morphology Data</article-title>. <source>J.&#x20;Appl. Cryst.</source> <volume>44</volume>, <fpage>1272</fpage>&#x2013;<lpage>1276</lpage>. <pub-id pub-id-type="doi">10.1107/S0021889811038970</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mounet</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gibertini</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schwaller</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Campi</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Merkys</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Marrazzo</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Two-Dimensional Materials from High-Throughput Computational Exfoliation of Experimentally Known Compounds</article-title>. <source>Nat. Nanotech</source> <volume>13</volume>, <fpage>246</fpage>&#x2013;<lpage>252</lpage>. <pub-id pub-id-type="doi">10.1038/s41565-017-0035-5</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nakamoto</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bellaiche</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Properties of Rare-Earth Iron Garnets from First Principles</article-title>. <source>Phys. Rev. B</source> <volume>95</volume>, <fpage>024434</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevB.95.024434</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Perdew</surname>
<given-names>J.&#x20;P.</given-names>
</name>
<name>
<surname>Ernzerhof</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Burke</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Rationale for Mixing Exact Exchange with Density Functional Approximations</article-title>. <source>J.&#x20;Chem. Phys.</source> <volume>105</volume>, <fpage>9982</fpage>&#x2013;<lpage>9985</lpage>. <pub-id pub-id-type="doi">10.1063/1.472933</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Phan</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Morales</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Chinnasamy</surname>
<given-names>C. N.</given-names>
</name>
<name>
<surname>Latha</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Harris</surname>
<given-names>V. G.</given-names>
</name>
<name>
<surname>Srikanth</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Magnetocaloric Effect in Bulk and Nanostructured Gd3Fe5O12 Materials</article-title>. <source>J.&#x20;Phys. D: Appl. Phys.</source> <volume>42</volume>, <fpage>115007</fpage>. <pub-id pub-id-type="doi">10.1088/0022-3727/42/11/115007</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rodic</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mitric</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tellgren</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rundlof</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kremenovic</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>True Magnetic Structure of the Ferrimagnetic Garnet Y3Fe5O12 and Magnetic Moments of Iron Ions</article-title>. <source>J.&#x20;Magnetism Magn. Mater.</source> <volume>191</volume>, <fpage>137</fpage>&#x2013;<lpage>145</lpage>. <pub-id pub-id-type="doi">10.1016/S0304-8853(98)00317-5</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rosen</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Notestein</surname>
<given-names>J.&#x20;M.</given-names>
</name>
<name>
<surname>Snurr</surname>
<given-names>R. Q.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Identifying Promising Metal-Organic Frameworks for Heterogeneous Catalysis via High-Throughput Periodic Density Functional Theory</article-title>. <source>J.&#x20;Comput. Chem.</source> <volume>40</volume>, <fpage>1305</fpage>&#x2013;<lpage>1318</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.25787</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sayetat</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>1986</year>). <article-title>Huge Magnetostriction in Tb3Fe5O12, Dy3Fe5O12, Ho3Fe5O12, Er3Fe5O12 Garnets</article-title>. <source>J.&#x20;Magnetism Magn. Mater.</source> <volume>58</volume>, <fpage>334</fpage>&#x2013;<lpage>346</lpage>. <pub-id pub-id-type="doi">10.1016/0304-8853(86)90456-7</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schlipf</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Betzinger</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Friedrich</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Le&#x17e;ai&#x107;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bl&#xfc;gel</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>HSE Hybrid Functional within the FLPAW Method and its Application to GdN</article-title>. <source>Phys. Rev. B</source> <volume>84</volume>, <fpage>125142</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevB.84.125142</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Serga</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Chumak</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Hillebrands</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>YIG Magnonics</article-title>. <source>J.&#x20;Phys. D: Appl. Phys.</source> <volume>43</volume>, <fpage>264002</fpage>. <pub-id pub-id-type="doi">10.1088/0022-3727/43/26/264002</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shavitt</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>1959</year>). <article-title>A Calculation of the Rates of the Ortho&#x2010;Para Conversions and Isotope Exchanges in Hydrogen</article-title>. <source>J.&#x20;Chem. Phys.</source> <volume>31</volume>, <fpage>1359</fpage>&#x2013;<lpage>1367</lpage>. <pub-id pub-id-type="doi">10.1063/1.1730599</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tabuchi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ishino</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Noguchi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ishikawa</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yamazaki</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Usami</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Coherent Coupling between a Ferromagnetic Magnon and a Superconducting Qubit</article-title>. <source>Science</source> <volume>349</volume>, <fpage>405</fpage>&#x2013;<lpage>408</lpage>. <pub-id pub-id-type="doi">10.1126/science.aaa3693</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="book">
<collab>Tm3Fe5O12 crystal structure</collab> (<year>2012</year>). <source>Datasheet from &#x201c;Pauling File Multinaries Edition &#x2013; 2012&#x201d; in Springermaterials</source> (<publisher-loc>Japan</publisher-loc>: <publisher-name>Springer-Verlag Berlin Heidelberg &#x26; Material Phases Data System (MPDS), Switzerland &#x26; National Institute for Materials Science NIMS</publisher-name>). <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://materials.springer.com/isp/crystallographic/docs/sd_0312594">https://materials.springer.com/isp/crystallographic/docs/sd_0312594</ext-link>. Copyright 2016</comment> . </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uchida</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Adachi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ohe</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Takahashi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ieda</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Spin Seebeck Insulator</article-title>. <source>Nat. Mater</source> <volume>9</volume>, <fpage>894</fpage>&#x2013;<lpage>897</lpage>. <pub-id pub-id-type="doi">10.1038/nmat2856</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vilela</surname>
<given-names>G. L. S.</given-names>
</name>
<name>
<surname>Abrao</surname>
<given-names>J.&#x20;E.</given-names>
</name>
<name>
<surname>Santos</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mendes</surname>
<given-names>J.&#x20;B. S.</given-names>
</name>
<name>
<surname>Rodr&#xed;guez-Su&#xe1;rez</surname>
<given-names>R. L.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Magnon-Mediated Spin Currents in Tm3Fe5O12/Pt with Perpendicular Magnetic Anisotropy</article-title>. <source>Appl. Phys. Lett.</source> <volume>117</volume>, <fpage>122412</fpage>. <pub-id pub-id-type="doi">10.1063/5.0023242</pub-id> </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vu</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Meisenheimer</surname>
<given-names>P. B.</given-names>
</name>
<name>
<surname>Heron</surname>
<given-names>J.&#x20;T.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Tunable Magnetoelastic Anisotropy in Epitaxial (111) Tm3Fe5O12 Thin Films</article-title>. <source>J.&#x20;Appl. Phys.</source> <volume>127</volume>, <fpage>153905</fpage>. <pub-id pub-id-type="doi">10.1063/1.5142856</pub-id> </citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wimmer</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Krakauer</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Weinert</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Freeman</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>1981</year>). <article-title>Full-Potential Self-Consistent Linearized-Augmented-Plane-Wave Method for Calculating the Electronic Structure of Molecules and surfaces: O2 molecule</article-title>. <source>Phys. Rev. B</source> <volume>24</volume>, <fpage>864</fpage>&#x2013;<lpage>875</lpage>. <pub-id pub-id-type="doi">10.1103/PhysRevB.24.864</pub-id> </citation>
</ref>
<ref id="B49">
<citation citation-type="book">
<collab>Y3Fe5O12 crystal structure</collab> (<year>2012</year>). <source>Datasheet from &#x201C;Pauling File Multinaries Edition &#x2013; 2012&#x201d; in Springermaterials</source> (<publisher-loc>Japan</publisher-loc>: <publisher-name>Springer-Verlag Berlin Heidelberg &#x26; Material Phases Data System (MPDS), Switzerland &#x26; National Institute for Materials Science NIMS</publisher-name>). <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://materials.springer.com/isp/crystallographic/docs/sd_0310635">https://materials.springer.com/isp/crystallographic/docs/sd_0310635</ext-link>. Copyright 2016</comment>. </citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Suram</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Shinde</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Newhouse</surname>
<given-names>P. F.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Solar Fuels Photoanode Materials Discovery by Integrating High-Throughput Theory and Experiment</article-title>. <source>Proc. Natl. Acad. Sci. USA</source> <volume>114</volume>, <fpage>3040</fpage>&#x2013;<lpage>3043</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1619940114</pub-id> </citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Garcia-Mendez</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Herbert</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Dudney</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Wolfenstine</surname>
<given-names>J.&#x20;B.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Elastic Properties of the Solid Electrolyte Li7La3Zr2O12 (LLZO)</article-title>. <source>Chem. Mater.</source> <volume>28</volume>, <fpage>197</fpage>&#x2013;<lpage>206</lpage>. <pub-id pub-id-type="doi">10.1021/acs.chemmater.5b03854</pub-id> </citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Donadio</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gygi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Galli</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>First Principles Simulations of the Infrared Spectrum of Liquid Water Using Hybrid Density Functionals</article-title>. <source>J.&#x20;Chem. Theor. Comput.</source> <volume>7</volume>, <fpage>1443</fpage>&#x2013;<lpage>1449</lpage>. <pub-id pub-id-type="doi">10.1021/ct2000952</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>