<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2023.1260583</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Learning with privileged and sensitive information: a gradient-boosting approach</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Yan</surname> <given-names>Siwen</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2379719/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Odom</surname> <given-names>Phillip</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Pasunuri</surname> <given-names>Rahul</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2424337/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Kersting</surname> <given-names>Kristian</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/321053/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Natarajan</surname> <given-names>Sriraam</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/399203/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Computer Science Department, University of Texas at Dallas</institution>, <addr-line>Dallas, TX</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Georgia Tech Research Institute, Georgia Institute of Technology</institution>, <addr-line>Atlanta, GA</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Amazon</institution>, <addr-line>Seattle, WA</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Computer Science, Hessian Center for AI (hessian.AI), Technical University of Darmstadt</institution>, <addr-line>Darmstadt</addr-line>, <country>Germany</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Thommen Karimpanal George, Deakin University, Australia</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Shahram Rahimi, Mississippi State University, United States; Manas Gaur, University of Maryland, Baltimore County, United States</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Siwen Yan <email>siwen.yan&#x00040;utdallas.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>13</day>
<month>11</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>6</volume>
<elocation-id>1260583</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>20</day>
<month>10</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2023 Yan, Odom, Pasunuri, Kersting and Natarajan.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Yan, Odom, Pasunuri, Kersting and Natarajan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license></permissions>
<abstract>
<p>We consider the problem of learning with sensitive features under the privileged information setting where the goal is to learn a classifier that uses features not available (or too sensitive to collect) at test/deployment time to learn a better model at training time. We focus on tree-based learners, specifically gradient-boosted decision trees for learning with privileged information. Our methods use privileged features as knowledge to guide the algorithm when learning from fully observed (usable) features. We derive the theory, empirically validate the effectiveness of our algorithms, and verify them on standard fairness metrics.</p></abstract>
<kwd-group>
<kwd>privileged information</kwd>
<kwd>fairness</kwd>
<kwd>gradient boosting</kwd>
<kwd>knowledge-based learning</kwd>
<kwd>sensitive features</kwd>
</kwd-group>
<contract-num rid="cn001">FA9550-19-1-039</contract-num>
<contract-sponsor id="cn001">Air Force Office of Scientific Research<named-content content-type="fundref-id">10.13039/100000181</named-content></contract-sponsor>
<counts>
<fig-count count="3"/>
<table-count count="4"/>
<equation-count count="12"/>
<ref-count count="53"/>
<page-count count="11"/>
<word-count count="8985"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1. Introduction</title>
<p>Machine learning methods that consider learning from sources beyond just a single set of labeled data have long been explored under several paradigms&#x02014;learning with advice (Towell and Shavlik, <xref ref-type="bibr" rid="B44">1994</xref>; Fung et al., <xref ref-type="bibr" rid="B16">2002</xref>; Maclin et al., <xref ref-type="bibr" rid="B34">2005</xref>; Kunapuli et al., <xref ref-type="bibr" rid="B28">2013</xref>; Das et al., <xref ref-type="bibr" rid="B9">2021</xref>), learning from preferences (Boutilier, <xref ref-type="bibr" rid="B3">2002</xref>; Drummond and Boutilier, <xref ref-type="bibr" rid="B13">2014</xref>; Pang et al., <xref ref-type="bibr" rid="B38">2018</xref>), learning from qualitative constraints (Altendorf et al., <xref ref-type="bibr" rid="B1">2005</xref>; Yang et al., <xref ref-type="bibr" rid="B50">2013</xref>; Kokel et al., <xref ref-type="bibr" rid="B26">2020</xref>), active learning (Settles, <xref ref-type="bibr" rid="B41">2012</xref>), transductive learning (Joachims, <xref ref-type="bibr" rid="B22">1999</xref>), and as knowledge injection inside deep learning (Ding et al., <xref ref-type="bibr" rid="B12">2018</xref>; Wang and Pan, <xref ref-type="bibr" rid="B48">2020</xref>; Bu and Cho, <xref ref-type="bibr" rid="B4">2021</xref>).</p>
<p>We view the problem of learning with sensitive information using the lens of privileged information. For instance, in a clinical study for improving adverse pregnancy outcomes, it is natural to solicit information about race or sexual orientation. While race can potentially affect the prior chances of an outcome (e.g., gestational diabetes or pre-term birth), the treatment plan in the clinic should not discriminate based on this feature. Similarly, while age/zipcode could be important to obtain a prior about the capacity to repay a loan, it should not be used as a feature (due to its sensitive nature) during deployment of the system. Chouldechova et al. (<xref ref-type="bibr" rid="B7">2018</xref>) consider these sensitive information as non-discriminatory for fair machine learning. Kilbertus et al. (<xref ref-type="bibr" rid="B24">2018</xref>) encrypt sensitive attributes, and Williamson and Menon (<xref ref-type="bibr" rid="B49">2019</xref>) measure the fairness risk on sensitive features.</p>
<p>In a different direction, Vapnik and Vashist (<xref ref-type="bibr" rid="B46">2009</xref>) introduced the problem of learning from privileged information where more information in the form of features is provided during training but is not available during testing/deployment. These <italic>privileged features</italic> could be sensitive features (race/age/sexual orientation) or features that are simply too expensive to collect during deployment (expensive sensors or FMRIs&#x02014;functional magnetic resonance imaging, in a small clinic). Hence, the classifier cannot use these privileged features during deployment but may still be able to use them to improve the quality of the model.</p>
<p>Our key idea is to use the privileged features as an &#x0201C;inductive bias&#x0201D; or as knowledge constraints. To this effect, we develop two versions of the gradient boosting algorithm&#x02014;in the first approach, a prior model is learned on the privileged/sensitive features. This prior model is then used to constrain the model learned from the fully observable features. Since this is inspired from the knowledge-based learning literature, we refer to this as <bold>KbPIB</bold> (<italic>knowledge-based privileged information boosting</italic>). In the second approach, the models over the privileged/sensitive features and the observed features are learned in a joint stage-wise manner. At each iteration, first a small tree is learned on the privileged features, the set of which is used as constraints for the observed feature model and the process is repeated. Since these are learned jointly, we refer to this as <bold>JPIB</bold> (<italic>joint privileged information boosting</italic>). The intuition is that while the privileged features provide extra information, they are not fully relied on when building the model. The resulting model, in essence, is a trade-off between the privileged information and the fully observed features&#x02014;as is typically done in advice-based methods where the data and the expert knowledge are explicitly considered when learning.</p>
<p>We make a few key contributions: First, inspired by Quadrianto and Sharmanska (<xref ref-type="bibr" rid="B40">2017</xref>), we pose the problem of fair machine learning with sensitive data using the framework of privileged information. Second, we present algorithms for learning trees <italic>via</italic> functional-gradient boosting and show the gradient updates. Specifically, we derive two different types of boosted algorithms that can effectively exploit the sensitive/privileged features. Finally, we perform exhaustive empirical evaluation that demonstrates the effectiveness of the proposed approaches on different types of test beds&#x02014;standard benchmark data sets, fairness data sets with sensitive information, and real-world medical data sets where the goals are to predict gestational diabetes, nephrotic syndrome, and rare disease occurrences. The results across data sets and evaluation metrics (including fairness metrics) clearly show the effectiveness of the algorithms.</p></sec>
<sec id="s2">
<title>2. Background</title>
<sec>
<title>2.1. Learning with privileged information</title>
<p>Learning with privileged information is inspired by richer forms of interactions between human teachers and students (Vapnik and Vashist, <xref ref-type="bibr" rid="B46">2009</xref>). Particular (labeled) examples are given to the student along with explanations and intuitions that are able to speed up the comprehension of novel concepts. More formally, learning with privileged information assumes that more information is known about the training examples. However, as the expert is not available for testing, this additional information is not available at test time. Thus, training examples have the form <inline-formula><mml:math id="M1"><mml:mrow><mml:mo>&#x02329;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:mrow><mml:mo>&#x0232A;</mml:mo></mml:mrow></mml:math></inline-formula> while testing examples have the form <inline-formula><mml:math id="M2"><mml:mrow><mml:mo>&#x02329;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:mrow><mml:mo>&#x0232A;</mml:mo></mml:mrow></mml:math></inline-formula>. <bold>CF</bold> refers to the classifier/normal features available during testing and <bold>PF</bold> refers to the privileged features.</p>
<p>Learning algorithms for privileged information have previously focused on SVMs (Vapnik and Vashist, <xref ref-type="bibr" rid="B46">2009</xref>; Sharmanska et al., <xref ref-type="bibr" rid="B42">2013</xref>). The original formulation&#x02014;SVM&#x0002B; (Vapnik and Vashist, <xref ref-type="bibr" rid="B46">2009</xref>)&#x02014;learned the difficulty of each training example. The key idea was to learn an SVM in the privileged space (using <inline-formula><mml:math id="M3"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo>&#x02329;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:mrow><mml:mo>&#x0232A;</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>) and find the margin with respect to this SVM for each training example. Training examples closer to the margin are considered &#x0201C;more difficult&#x0201D; as they are closer to the decision boundary while examples farther from the margin are considered &#x0201C;less difficult.&#x0201D; Since the introduction of the new learning paradigm and the corresponding SVM&#x0002B; approach, there is a growing body of work on learning with privileged information. Pechyony and Vapnik (<xref ref-type="bibr" rid="B39">2010</xref>) developed a theoretical justification of the learning setting. Liang et al. (<xref ref-type="bibr" rid="B31">2009</xref>) established links between the SVM&#x0002B; and the multi-task learning. Hern&#x000E1;ndez-Lobato et al. (<xref ref-type="bibr" rid="B20">2014</xref>) showed that the privileged information can naturally be treated as noise in the latent function of a Gaussian process classifier (GPC). In contrast to the standard GPC setting, the latent function becomes a natural measure of confidence about the training data by modulating the slope of the GPC sigmoid likelihood function.</p>
<p>Most closely related to our study, Chen et al. (<xref ref-type="bibr" rid="B5">2012</xref>) extend the setting to AdaBoost, and Lapin et al. (<xref ref-type="bibr" rid="B30">2014</xref>) relate privileged information to importance weighting within SVMs. Decision tree learners, however, have not been explored in this context yet. Instead of giving more importance to certain examples, we establish a novel connection to knowledge-based machine learning that relies on existing knowledge (Section 3.2). We show that the knowledge we have beforehand, which can be described with privileged features, can also be represented using labels assigned to each training example. These labels help guide the learning process. Moreover, we improve the learning process by introducing a regularization term into the log-likelihood for boosting method. This regularization term is calculated as the KL divergence between the distribution using classifier features and the distribution using privileged features, which are available only at training time and not during testing.</p>
<p>While our setting is similar to the generalized distillation (Lopez-Paz et al., <xref ref-type="bibr" rid="B32">2016</xref>), the fundamental principles are different. We focus on two sets of features&#x02014;privileged features and normal features. Our goal is to build a model (teacher model) on privileged features that guides the learning of a model (student model) on normal features to improve performance. While our strategy at a high-level appears similar to knowledge distillation by Hinton et al. (<xref ref-type="bibr" rid="B21">2015</xref>), there are notable and important differences as follows: (1) the teacher function and student function are learned sequentially, and (2) predictions of teacher model are included as soft labels for the student model.</p>
<p>Our study is also related to knowledge injection in deep networks. Ding et al. (<xref ref-type="bibr" rid="B12">2018</xref>) use mean images as (color) knowledge to produce class weight and object occurrence frequencies as scene knowledge to determine scene weight; Wang and Pan (<xref ref-type="bibr" rid="B48">2020</xref>) integrate logical knowledge in the form of first-order logic as knowledge regularization into deep learning system; Bu and Cho (<xref ref-type="bibr" rid="B4">2021</xref>) perform neuro-symbolic integration using domain knowledge as first-order logic rules.</p></sec>
<sec>
<title>2.2. Functional gradient boosting</title>
<p>Many probabilistic learning methods learn the conditional distribution <italic>P</italic>(<italic>y</italic><sub><italic>i</italic></sub>|<bold>x</bold><sub><italic>i</italic></sub>; &#x003C8;) using standard techniques such as gradient-descent that is usually performed on the log-likelihood w.r.t. parameters to find the best set of parameters that model the training data. Functional Gradient Boosting methods (GB; Friedman, <xref ref-type="bibr" rid="B15">2001</xref>; Dietterich et al., <xref ref-type="bibr" rid="B11">2008</xref>; Natarajan et al., <xref ref-type="bibr" rid="B37">2012</xref>, <xref ref-type="bibr" rid="B36">2015</xref>), on the other hand, represent the likelihood in a functional form (typically using the sigmoid function) <inline-formula><mml:math id="M4"><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:munder><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:math></inline-formula> where &#x003C8; is a regression function defined over the examples. Given this representation, GB methods obtain the gradient of the log-likelihood w.r.t. &#x003C8;(<italic>y</italic><sub><italic>i</italic></sub> &#x0003D; 1, <bold>x</bold><sub><italic>i</italic></sub>) for each training example, <italic>x</italic><sub><italic>i</italic></sub> as: &#x00394;(<italic>y</italic><sub><italic>i</italic></sub>) &#x0003D; <italic>I</italic>(<italic>y</italic><sub><italic>i</italic></sub> &#x0003D; 1)&#x02212;<italic>P</italic>(<italic>y</italic><sub><italic>i</italic></sub> &#x0003D; 1|<bold>x</bold><sub><italic>i</italic></sub>; &#x003C8;), where <italic>I</italic> is an indicator function which returns 1 for positive examples and 0 for negative examples in a binary classification task. The GB approach starts with an initial regression function, &#x003C8;<sub>0</sub> &#x0003D; 0 to compute the probabilities of the training examples and thereby the gradients &#x00394;<sub>1</sub>. A regression function (typically a tree), <inline-formula><mml:math id="M5"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, is fit on the training examples with the gradients as the target regression values. This learned function is now added to &#x003C8;<sub>0</sub>, and the process is repeated with <inline-formula><mml:math id="M6"><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. Given that the stage-wise growth of trees resembles boosting, and that the process involves computing gradients of functions, this method is called <italic>Gradient Boosting</italic> (GB).</p></sec>
<sec>
<title>2.3. Sensitive features in fairness</title>
<p>Several studies within fairness in ML treat the sensitive information (e.g., race, gender, or financial status) as non-discriminatory (&#x0017D;liobait&#x00117;, <xref ref-type="bibr" rid="B53">2017</xref>; Chouldechova et al., <xref ref-type="bibr" rid="B7">2018</xref>). Several approaches have been proposed to avoid the use of sensitive features, including by utilizing encrypted sensitive attributes (Choudhuri et al., <xref ref-type="bibr" rid="B6">2017</xref>; Kilbertus et al., <xref ref-type="bibr" rid="B24">2018</xref>) or utilizing sensitive features to measure the fairness risk by proposing a new definition of fairness to include categorical or real-valued sensitive groups beyond binary sensitive features (Angwin et al., <xref ref-type="bibr" rid="B2">2016</xref>; Williamson and Menon, <xref ref-type="bibr" rid="B49">2019</xref>). Krasanakis et al. (<xref ref-type="bibr" rid="B27">2018</xref>) reweigh training samples on trade-offs between accuracy and disparate impact. Kamishima et al. (<xref ref-type="bibr" rid="B23">2012</xref>) regularize on prejudice (a statistical dependence between sensitive features and other information) to achieve fairness. Quadrianto and Sharmanska (<xref ref-type="bibr" rid="B40">2017</xref>) enforce fairness constraints through privileged learning. They consider the setting from the study by Vapnik and Vashist (<xref ref-type="bibr" rid="B46">2009</xref>) to build a privileged model on all features, optimizing the prediction boundary of a privileged model and adapting the boundary of the normal model. On the other hand, we use a boosted model as the privileged model relying only on the privileged features and incorporate constraints from the privileged model into the objective of the model constructed on non-privileged features. Wang et al. (<xref ref-type="bibr" rid="B47">2021</xref>) approach fairness by putting strict restraints on the ability to infer sensitive features from the available features. Their approach focuses on improving fairness while maintaining performance. Alternatively, our approach aims to leverage the sensitive information to improve performance. Empirically, we demonstrate that our approach maintains fairness.</p></sec></sec>
<sec id="s3">
<title>3. Boosting with privileged sensitive information</title>
<p><bold>Motivating real-world task</bold>: The Nulliparous Pregnancy Outcomes Study (NuMoM2b) monitors expectant mothers with the goal of predicting adverse pregnancy outcomes (Haas et al., <xref ref-type="bibr" rid="B18">2015</xref>). The data set includes clinical tests (e.g., BMI, METs) and demographic information. Our goal is to use this data to predict gestational diabetes. While the prevalence of gestational diabetes varies significantly across ethnic groups, it may not be appropriate to use the sensitive demographic information to make the diagnoses. Thus, we may utilize this privileged information during training but want to withhold it from our diagnostic models. A similar consideration is in our rare disease data where certain demographic information, such as age, gender, and marital status, is considered privileged and cannot be used during deployment. While we focus on four specific medical tasks, one could imagine such situations in other high social impact problems including, but not limited to, credit card/home/auto/education loan approvals, hiring decisions, clinical study recruitment, or allocation of resources, where some sensitive information could be used while training to better understand the problem but cannot be used during deployment.</p>
<sec>
<title>3.1. Problem formulation</title>
<p>Recalling that our goal is to learn robust models that do not include sensitive/privileged information but still leverage them to improve training. Our problem is formally defined as follows:</p>
<boxed-text>
<p><sc><bold>Given:</bold></sc> A set of training examples <inline-formula><mml:math id="M7"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo>&#x02329;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:mrow><mml:mo>&#x0232A;</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula> and a set of test examples <inline-formula><mml:math id="M8"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo>&#x02329;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:mrow><mml:mo>&#x0232A;</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, where</p>
<disp-formula id="E1"><mml:math id="M9"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle><mml:mo>&#x0222A;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle><mml:mtext>&#x02003;</mml:mtext><mml:mi>&#x00026;</mml:mi><mml:mtext>&#x02003;</mml:mtext><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle><mml:mo>&#x02229;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mi>&#x02205;</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><sc><bold>To Do:</bold></sc> Learn a classifier that employs only the classifier features <bold>CF</bold> for classifying the test data and can utilize the privileged features <bold>PF</bold> effectively in learning a better model.</p>
</boxed-text>
<p><bold>F</bold> is the set of all features, <bold>CF</bold> is the set of features that are available at both training and testing time (and we call them <italic>classifier features</italic>), <bold>PF</bold> are the privileged features that are accessible only during training and not during testing, <italic>y</italic><sub><italic>i</italic></sub> is the label of the <italic>ith</italic> example and <bold>x</bold><sub><italic>i</italic></sub> is the feature of that example. We use {} to denote sets. For example, the input to the algorithm is the set of all examples <inline-formula><mml:math id="M10"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo>&#x02329;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:mrow><mml:mo>&#x0232A;</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>. We first consider a knowledge-based approach to leverage with privileged features. Then, we extend this approach to joint training over the classifier and the privileged features. While we use tree-based classifiers, our approach can easily be extended to other clustering/classification techniques.</p></sec>
<sec>
<title>3.2. Knowledge-based privileged information boosting</title>
<p>Inspired by knowledge-based machine learning methods, Fung et al. (<xref ref-type="bibr" rid="B16">2002</xref>), for example, reformulated SVM classifier that uses previous knowledge in the form of multiple polyhedral sets; Kunapuli et al. (<xref ref-type="bibr" rid="B28">2013</xref>) incorporate expert advice in states and actions by stating preferences; Towell and Shavlik (<xref ref-type="bibr" rid="B44">1994</xref>) map domain theories in propositional logic into neural networks, which leverage external knowledge <italic>via</italic> human input to guide the learning process; we consider privileged information as a source of high-quality knowledge. We introduce two models: a model learned over the classifier features and a privileged model that is learned over the privileged features in <xref ref-type="table" rid="T5">Algorithm 1</xref>. By attempting to guide the predictions of the classifier model with the privileged model, we can potentially find a way to insert informed priors to the labels based on both the privileged and classifier features. We first train a model over privileged features at line 2 in <xref ref-type="table" rid="T5">Algorithm 1</xref>. Then, from lines 3 to 9, we learn a model over the classifier features while reducing the margin with the privileged model. An overview of our <bold>KbPIB</bold> approach is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<table-wrap position="float" id="T5">
<label>Algorithm 1</label>
<caption><p><bold>KbPIB</bold>: <underline>K</underline>nowledge-<underline>b</underline>ased <underline>P</underline>rivileged <underline>I</underline>nformation <underline>B</underline>oosting.</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr><td align="left" valign="top"><bold>Input</bold>: Classifier features: training data <inline-formula><mml:math id="M11"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>; validation data <inline-formula><mml:math id="M12"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>; privileged features: training data <inline-formula><mml:math id="M13"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>; validation data <inline-formula><mml:math id="M14"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula></td></tr>
<tr><td align="left" valign="top"><bold>Parameter</bold>: Number of trees <italic>N</italic>, early-stop parameter <italic>P</italic><bold>Output</bold>: Learned model &#x003C8;</td></tr>
<tr><td align="left" valign="top"><monospace>1: Initialize model &#x003C8;<sub>0</sub> &#x0003D; 0, counter <italic>C</italic> &#x0003D; 0, score <italic>R</italic>, best number of trees index <italic>j</italic></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>2: &#x003C8;<sup><bold>PF</bold></sup>&#x02190; <bold>NF</bold>(<inline-formula><mml:math id="M15"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>) { <xref ref-type="supplementary-material" rid="SM1">Supplementary Algorithm 1</xref>}</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>3: <bold>for</bold> <italic>i</italic> &#x0003D; 1 <bold>to</bold> <italic>N</italic> <bold>do</bold></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>4: &#x00394;<sub><italic>i</italic></sub>&#x02190; ComputeGradient(<inline-formula><mml:math id="M16"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msup></mml:math></inline-formula>) {Equation (2)}</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>5: <inline-formula><mml:math id="M17"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo></mml:math></inline-formula> FitRegressionValue(<inline-formula><mml:math id="M18"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>6: <inline-formula><mml:math id="M19"><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>7: <italic>R</italic><sub>val</sub> &#x02190; Evaluate(<inline-formula><mml:math id="M20"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>8: <italic>j</italic>, <italic>R</italic>, <italic>C</italic> &#x02190; EarlyStop(<italic>i</italic>, <italic>j</italic>, <italic>R</italic>, <italic>R</italic><sub>val</sub>, <italic>C</italic>, <italic>P</italic>) { <xref ref-type="supplementary-material" rid="SM1">Supplementary Algorithm 2</xref>}</monospace> </td></tr>
<tr><td align="left" valign="top"><monospace>9: <bold>end for</bold></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>10: <bold>return</bold> &#x003C8;<sub><italic>j</italic></sub></monospace></td></tr>
</tbody>
</table>
</table-wrap>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Overview of proposed approaches KbPIB and JPIB [Both approaches train a model on CF, while KbPIB trains a model on PF once and uses it as a bias for the classifier, JPIB learns models on CF and PF in an iterative manner (in a manner loosely similar to co-training); models on PF data are dropped after training to avoid consideration during testing/deployment].</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-06-1260583-g0001.tif"/>
</fig>
<p>In Vapnik&#x00027;s SVM&#x0002B; model, the <bold>PF</bold> features were used to define an oracle function that can predict the slack on each example. In our probabilistic framework, we use the <bold>PF</bold> features to build an oracle model that can predict a close approximation to the true distribution of each example which is not captured by the discrete class labels. Instead of modeling the error (distance between the labels and the underlying distribution), we directly model the distribution of labels using privileged features during training. We use <italic>P</italic>(<italic>y</italic>|<bold>x</bold><sup><bold>PF</bold></sup>; &#x003C8;&#x02032;) to indicate this true label distribution learned over privileged features and <italic>P</italic>(<italic>y</italic>|<bold>x</bold><sup><bold>CF</bold></sup>; &#x003C8;) for the distribution learned over classifier features. Similar to SVM&#x0002B;, we can now use the privileged features to model this difference between the true distribution and the label distribution. Since the training labels are completely observed, the privileged features can be directly used to model the distribution of the examples. Thus, we learn a model that minimizes the error of the model over the training labels and the margin between the distribution <italic>P</italic>(<italic>y</italic>|<bold>x</bold><sup><bold>PF</bold></sup>; &#x003C8;&#x02032;) and <italic>P</italic>(<italic>y</italic>|<bold>x</bold><sup><bold>CF</bold></sup>; &#x003C8;),</p>
<disp-formula id="E2"><mml:math id="M21"><mml:msub><mml:mi>min</mml:mi><mml:mi>&#x003C8;</mml:mi></mml:msub><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mi>i</mml:mi></mml:munder><mml:mo stretchy='false'>(</mml:mo></mml:mstyle><mml:munder><mml:munder><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>log</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo stretchy='true'>&#x0FE38;</mml:mo></mml:munder><mml:mrow><mml:mtext>NLL</mml:mtext></mml:mrow></mml:munder><mml:mo>+</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:munder><mml:munder><mml:mrow><mml:mtext>KL</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo stretchy='true'>&#x0FE38;</mml:mo></mml:munder><mml:mrow><mml:mtext>KLDivergence</mml:mtext></mml:mrow></mml:munder><mml:mo stretchy='false'>)</mml:mo></mml:math></disp-formula>
<p>NLL denotes the negative log-likelihood of the training data that models the error while KL denotes the KL divergence between <italic>P</italic>(&#x0002A;;&#x003C8;&#x02032;) and <italic>P</italic>(&#x0002A;;&#x003C8;) and is equal to <inline-formula><mml:math id="M22"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo class="qopname">log</mml:mo><mml:mfrac><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:math></inline-formula>. We use &#x003B1; to model the trade-off between fitting to the labeled data versus fitting to the distribution learned over the privileged features. We can now use gradient boosting with respect to <inline-formula><mml:math id="M23"><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> to minimize this objective function.</p>
<p>Notably, in our formulation, the model &#x003C8;&#x02032; could be provided by the domain expert on the privileged features (for instance, a Bayesian network or a neural network that is used in the literature on these privileged features). We do not assume any specific form for &#x003C8;&#x02032;, and the goal is to use this privileged knowledge. In our experiments, we learn &#x003C8;&#x02032; from data. If the model of &#x003C8;&#x02032; is provided, one could treat that as a regularizer (similar to knowledge-based learning).</p>
<p>The first term of our objective function is the standard log-likelihood function which has the gradient as follows<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref>:</p>
<disp-formula id="E3"><label>(1)</label><mml:math id="M50"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mo class="qopname">log</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mi>I</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>For the second term, we derive the gradients below:</p>
<disp-formula id="E4"><mml:math id="M26"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mtext>KL</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>log</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>log</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mi>log</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mfrac><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mi>log</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>+</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mfrac><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mi>log</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x000B7;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>+</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x000B7;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>We combine the gradient terms to get the final gradient for each example as follows<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref>:</p>
<disp-formula id="E5"><label>(2)</label><mml:math id="M27"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mi>&#x00394;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mi>I</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Intuitively, if the learned distribution has a higher probability of an example belonging to the positive class compared with the distribution, <inline-formula><mml:math id="M29"><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> would be positive and the gradient would be pushed lower. Hence, the additional term would push the gradient (weighted by &#x003B1;) toward the distribution as predicted by our privileged features.</p>
<p>The parameter &#x003B1; controls the influence of the privileged data on the learned distribution. When &#x003B1; &#x0003D; 0, privileged features are ignored resulting in the standard functional gradient. As &#x003B1; is increased, the gradient is pushed lower, for example, where predicted probability is higher than true probability (w.r.t. privileged model) and vice versa.</p>
</sec>
<sec>
<title>3.3. Joint privileged information boosting</title>
<p>While the previous approach used the privileged information to influence final model learned over <bold>CF</bold> at each step in gradient boosting, it did not leverage this learned model to further tune the privileged tree labels. By attempting to reduce the margin by jointly training the two models, we can potentially find more consistent predictions based on both the privileged and classifier features. Similar to Equation (2), the gradients can be computed for learning the true distribution using the privileged features with <italic>P</italic>(&#x0002A;;&#x003C8;) and <italic>P</italic>(&#x0002A;;&#x003C8;&#x02032;) switched around.</p>
<disp-formula id="E7"><label>(3)</label><mml:math id="M44"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mi>&#x00394;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mo stretchy='false'>[</mml:mo><mml:mi>I</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>]</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mo stretchy='false'>[</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mi>i</mml:mi><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>C</mml:mi><mml:mi>F</mml:mi></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo stretchy='false'>]</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Given these gradients, we can now describe our approach called <bold>JPIB</bold> to perform gradient boosting jointly over the classifier features and the privileged information. We iteratively learn regression functions (trees in our case) to fit to these gradients. However, the key difference from <bold>KbPIB</bold> is that we perform co-ordinate gradient descent, i.e., we alternate between taking a gradient step along &#x003C8; and &#x003C8;&#x02032;. From lines 2 to 11 in <xref ref-type="table" rid="T6">Algorithm 2</xref>, we learn one regression tree using the gradients based on the classifier features (lines 2&#x02013;5), compute the gradients for the privileged features, learn a tree for the privileged features (lines 8&#x02013;10), and repeat this at most <italic>N</italic> times to generate at most <italic>N</italic> trees of the boosting model. The early-stop mechanism at line 7 helps return the best performing model on validation data (line 6).</p>
<table-wrap position="float" id="T6">
<label>Algorithm 2</label>
<caption><p><bold>JPIB</bold>: <underline>J</underline>oint <underline>P</underline>rivileged <underline>I</underline>nformation <underline>B</underline>oosting.</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr><td align="left" valign="top"><bold>Input</bold>: Classifier features: training data <inline-formula><mml:math id="M30"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>, validation data <inline-formula><mml:math id="M31"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>; privileged features: training data <inline-formula><mml:math id="M32"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula></td></tr>
<tr><td align="left" valign="top"><bold>Parameter</bold>: Number of trees <italic>N</italic>, early-stop patience <italic>P</italic><bold>Output</bold>: Learned model &#x003C8;</td></tr>
<tr><td align="left" valign="top"><monospace>1: Initialize models <inline-formula><mml:math id="M33"><mml:msubsup><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> and &#x003C8;<sub>0</sub> &#x0003D; 0, counter <italic>C</italic> &#x0003D; 0, score <italic>R</italic>, best number of trees index <italic>j</italic></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>2: <bold>for</bold> <italic>i</italic> &#x0003D; 1 <bold>to</bold> <italic>N</italic> <bold>do</bold></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>3: &#x00394;<sub><italic>i</italic></sub>&#x02190; ComputeGradient(<inline-formula><mml:math id="M34"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:math></inline-formula>) {Equation (2)}</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>4: <inline-formula><mml:math id="M35"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo></mml:math></inline-formula> FitRegressionValue(<inline-formula><mml:math id="M36"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>5: <inline-formula><mml:math id="M37"><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>6: <italic>R</italic><sub>val</sub> &#x02190; Evaluate(<inline-formula><mml:math id="M38"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>val</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>7: <italic>j</italic>, <italic>R</italic>, <italic>C</italic> &#x02190; EarlyStop(<italic>i</italic>, <italic>j</italic>, <italic>R</italic>, <italic>R</italic><sub>val</sub>, <italic>C</italic>, <italic>P</italic>) {<xref ref-type="supplementary-material" rid="SM1">Supplementary Algorithm 2</xref>}</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>8: <inline-formula><mml:math id="M39"><mml:msubsup><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>&#x02190;</mml:mo></mml:math></inline-formula> ComputeGradient(<inline-formula><mml:math id="M40"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) {Equation (3)}</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>9: <inline-formula><mml:math id="M41"><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>&#x02190;</mml:mo></mml:math></inline-formula> FitRegressionValue(<inline-formula><mml:math id="M42"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:math></inline-formula>)</monospace></td></tr>
<tr><td align="left" valign="top"><monospace>10: <inline-formula><mml:math id="M43"><mml:msubsup><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>&#x02190;</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:math></inline-formula></monospace> </td></tr>
<tr><td align="left" valign="top"><monospace>11: <bold>end for</bold></monospace></td></tr>
<tr><td align="left" valign="top"><monospace>12: <bold>return</bold> &#x003C8;<sub><italic>j</italic></sub></monospace></td></tr>
</tbody>
</table>
</table-wrap></sec>
<sec>
<title>3.4. Sensitive attributes and fairness constraints</title>
<p>Notably, since our algorithms drop the privileged information after learning, one could argue that they do not discriminate between the different groups at deployment time. However, one could go even deeper and establish <bold>a strong connection between the learning framework and the fairness constraints</bold>. Given the above definitions of the objective function, several fairness constraints can be easily captured by our model. For instance, to handle <bold>metric fairness</bold>, the privileged model could simply be a constraint of the form</p>
<disp-formula id="E9"><mml:math id="M46"><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>&#x02200;</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mspace width="2.77695pt" class="tmspace"/><mml:mo>&#x021D2;</mml:mo><mml:mspace width="2.77695pt" class="tmspace"/><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>that can be used inside the second term of Equations (2) and (3), where the second term is the probability of the constraint satisfied by the model (computed by counting). <bold>Weakly meritocratic fairness</bold> can be handled by the form</p>
<disp-formula id="E10"><mml:math id="M47"><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>&#x02200;</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02265;</mml:mo><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mspace width="2.77695pt" class="tmspace"/><mml:mo>&#x021D2;</mml:mo><mml:mspace width="2.77695pt" class="tmspace"/><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0003E;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>while <bold>group fairness</bold> can be handled by</p>
<disp-formula id="E11"><mml:math id="M48"><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>g</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02227;</mml:mo><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>g</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mspace width="2.77695pt" class="tmspace"/><mml:mo>&#x021D2;</mml:mo><mml:mspace width="2.77695pt" class="tmspace"/><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>and <bold>group parity</bold> can be handled by using precision and recall. Similar to the metric constraints, all these constraints can be included in the second term of the model. Essentially one could drop the privileged tree model and use these constraints. Another way is to include these constraints along with the privileged model. However, as we show in our experiments, with treating the sensitive attributes as privileged features, the algorithm performs significantly better in terms of the fairness criteria compared with the boosting baseline.</p></sec></sec>
<sec id="s4">
<title>4. Experiments</title>
<p>Our experimental evaluations aim to answer the following questions:</p>
<list list-type="simple">
<list-item><p><bold>Q1:</bold> How effective is incorporating privileged information into gradient boosting?</p></list-item>
<list-item><p><bold>Q2:</bold> Can jointly updating the privileged model with the classifier improve performance?</p></list-item>
<list-item><p><bold>Q3:</bold> How is model fairness affected by withholding sensitive information from the classifier?</p></list-item>
</list>
<p>We present empirical evaluations of our proposed approaches&#x02014;(<bold>KbPIB</bold>) and (<bold>JPIB</bold>). We evaluate the approaches in two ways. To evaluate the effect of privileged information, we compare against learning a gradient-boosted model over only the classifier/normal features, <bold>NF</bold>. To evaluate fairness, our approaches are compared with <bold>All</bold>, which is learned over both <bold>CF</bold> and imputed <bold>PF</bold> based on mode. Notably, though we explicitly evaluate against the SVM-based approach and fairness approach, the key question in our study is whether the notion of privileged information can help gradient boosting and whether the sensitive features are handled appropriately. We adopt 10-fold cross-validation for all datasets: 8-folds of training, 1-fold of validation, and 1-fold of test. Due to the data size and very few negative instances in the dataset Nephrotic Syndrome, we use 5-fold cross-validation: 3-folds of training, 1-fold of validation, and 1-fold of test. The value of &#x003B1; and thresholds of precision and recall are selected based on the validation data. More details are presented in <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>. The experiments are conducted on the machine with CentOS Linux 7, CPU of Intel Xeon E5-2630 with 2.40 GHz and 16 cores, and 512 GB RAM. The source code (details of dependency) of our methods and prepared data can be downloaded.<xref ref-type="fn" rid="fn0003"><sup>3</sup></xref></p>
<sec>
<title>4.1. Datasets</title>
<p>We employ three types of datasets: standard benchmarks, medical datasets, and fairness benchmarks. The standard benchmarks consist of three datasets from UCI ML repository (Dheeru and Taniskidou, <xref ref-type="bibr" rid="B10">2017</xref>). The fairness benchmarks include 12 datasets from 10 data sources: Adult (Kohavi, <xref ref-type="bibr" rid="B25">1996</xref>), Diabetes (Diab.) (Strack et al., <xref ref-type="bibr" rid="B43">2014</xref>), Dutch Census (Dutch) (Van der Laan, <xref ref-type="bibr" rid="B45">2000</xref>), Bank Marketing (Bank) (Moro et al., <xref ref-type="bibr" rid="B35">2014</xref>), Credit Card Clients (Credit) (Yeh and Lien, <xref ref-type="bibr" rid="B51">2009</xref>), COMPAS (COMP.) and COMPAS Violence (C. V.) (Angwin et al., <xref ref-type="bibr" rid="B2">2016</xref>), Student&#x02013;Mathematics (St. M.) and Student&#x02013;Portuguese (St. P.) (Cortez and Silva, <xref ref-type="bibr" rid="B8">2008</xref>), OULAD (OUL.) (Kuzilek et al., <xref ref-type="bibr" rid="B29">2017</xref>), Communities and Crime (Comm.), and KDD Census Income (KDD) (Dheeru and Taniskidou, <xref ref-type="bibr" rid="B10">2017</xref>). While we describe the medical datasets in more detail, the properties of standard and fairness benchmarks are presented in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Standard benchmark datasets and fairness benchmark datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="left"><bold>PF</bold></th>
<th valign="top" align="center"><bold>&#x00023;F</bold></th>
<th valign="top" align="center"><bold>&#x00023;Instances</bold></th>
<th valign="top" align="center"><bold>N/P</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Heart</td>
<td valign="top" align="left">Tests</td>
<td valign="top" align="center">13</td>
<td valign="top" align="center">297</td>
<td valign="top" align="center">1.17</td>
</tr> <tr>
<td valign="top" align="left">Car</td>
<td valign="top" align="left">Main.</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">1,728</td>
<td valign="top" align="center">2.34</td>
</tr> <tr>
<td valign="top" align="left">Spam</td>
<td valign="top" align="left">Word freq.</td>
<td valign="top" align="center">57</td>
<td valign="top" align="center">4,601</td>
<td valign="top" align="center">1.54</td>
</tr> <tr>
<td valign="top" align="left">Adult</td>
<td valign="top" align="left">Age, race, sex</td>
<td valign="top" align="center">13</td>
<td valign="top" align="center">30,162</td>
<td valign="top" align="center">3.02</td>
</tr> <tr>
<td valign="top" align="left">Diab.</td>
<td valign="top" align="left">Sex</td>
<td valign="top" align="center">17</td>
<td valign="top" align="center">46,176</td>
<td valign="top" align="center">3.13</td>
</tr> <tr>
<td valign="top" align="left">Dutch</td>
<td valign="top" align="left">Sex</td>
<td valign="top" align="center">11</td>
<td valign="top" align="center">60,420</td>
<td valign="top" align="center">1.10</td>
</tr> <tr>
<td valign="top" align="left">Bank</td>
<td valign="top" align="left">Age, mar.</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">45,211</td>
<td valign="top" align="center">7.55</td>
</tr> <tr>
<td valign="top" align="left">Credit</td>
<td valign="top" align="left">Edu., mar., sex</td>
<td valign="top" align="center">23</td>
<td valign="top" align="center">30,000</td>
<td valign="top" align="center">3.52</td>
</tr> <tr>
<td valign="top" align="left">COMP.</td>
<td valign="top" align="left">Race, sex</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">6,172</td>
<td valign="top" align="center">1.20</td>
</tr> <tr>
<td valign="top" align="left">C. V.</td>
<td valign="top" align="left">Race, sex</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">4,015</td>
<td valign="top" align="center">5.16</td>
</tr> <tr>
<td valign="top" align="left">Comm.</td>
<td valign="top" align="left">Race</td>
<td valign="top" align="center">21</td>
<td valign="top" align="center">1,994</td>
<td valign="top" align="center">15.34</td>
</tr> <tr>
<td valign="top" align="left">St. M.</td>
<td valign="top" align="left">Age, sex</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">395</td>
<td valign="top" align="center">0.49</td>
</tr> <tr>
<td valign="top" align="left">St. P.</td>
<td valign="top" align="left">Age, sex</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">649</td>
<td valign="top" align="center">0.18</td>
</tr> <tr>
<td valign="top" align="left">OUL.</td>
<td valign="top" align="left">Sex</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">21,562</td>
<td valign="top" align="center">0.47</td>
</tr> <tr>
<td valign="top" align="left">KDD</td>
<td valign="top" align="left">Race, sex</td>
<td valign="top" align="center">23</td>
<td valign="top" align="center">284,556</td>
<td valign="top" align="center">15.35</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p><bold>PF</bold>, privileged features; &#x00023;<bold>F</bold>, &#x00023;features; N/P, negative positive ratio; main., maintenance; edu., education; mar., marital status.</p>
</table-wrap-foot>
</table-wrap></sec>
<sec>
<title>4.2. Real-world medical datasets</title>
<sec>
<title>4.2.1. NuMoM2b_a</title>
<p>Polygenic risk scores (PRS) for type 2 diabetes (T2D) can improve risk prediction for gestational diabetes (GD) (Haas et al., <xref ref-type="bibr" rid="B18">2015</xref>). We use PRS as the privileged feature. Demographic information and clinical history serve as normal features: body mass index (BMI), exercise levels or metabolic equivalents of time (METs), age, diabetes history (DM_Hist), polycystic ovary syndrome (PCOS), and high blood pressure (HiBP). The classification task is to predict GD. There are 3,657 instances with Neg/Pos ratio of 25.89.</p></sec>
<sec>
<title>4.2.2. NuMoM2b_b</title>
<p>We use the attribute race as privileged feature, which often is not usable during test or deployment for privacy concern (Haas et al., <xref ref-type="bibr" rid="B18">2015</xref>). The normal features and classification task are same as NuMoM2b_a. There are 6,164 instances with the Neg/Pos ratio of 23.76.</p></sec>
<sec>
<title>4.2.3. Nephrotic syndrome</title>
<p>A novel dataset of symptoms that indicates kidney damage is sourced from Dr Lal PathLabs, India.<xref ref-type="fn" rid="fn0004"><sup>4</sup></xref> This consists of 50 clinical reports with patient history information. The privileged features are age and gender. History of other diseases, Edema duration, urine test, and blood reports are used as normal features. The classification task is to predict Nephrotic Syndrome. The Neg/Pos ratio is 0.14.</p></sec>
<sec>
<title>4.2.4. Rare disease</title>
<p>This dataset is collected to identify rare diseases from behavioral data (MacLeod et al., <xref ref-type="bibr" rid="B33">2016</xref>). We consider age, gender, and marital status as privileged features. The survey questions are used as normal features and include demographic information, disease information, technology use, and health care professional inputs. The boolean classification task is to predict the presence of rare diseases. There are 284 instances with the Neg/Pos ratio of 2.69 and 69 features.</p></sec></sec>
<sec>
<title>4.3. Results</title>
<p>We first compare our <bold>KbPIB</bold> and <bold>JPIB</bold> approaches to the baseline <bold>NF</bold> that does not use privileged information during training. We evaluate the approaches based on the AUC ROC, as shown in <xref ref-type="table" rid="T2">Table 2</xref>, due to class imbalance. Blue denotes when either of our approaches outperform the baseline. The best performance is bolded. Overall, our approaches outperform the baseline, showing improvement in 18 out of 19 datasets. Both of our methods perform at least as well as the baseline across the rest of the datasets. Notably, both <bold>KbPIB</bold> (2 out of 4) and <bold>JPIB</bold> (3 out of 4) outperform the baseline in real-world medical tasks, where sensitive information includes demographic information. The NS dataset, on the other hand, has a large number of positives to negatives (but a small number of examples over all), and the base model that uses the urine tests gets nearly perfect example. It is an example of a situation where privileged information does not quite helpful, and it is natural that in many domains, the data might be sufficient to learn a good predictive model and extra information may not be helpful. We present this result to show the absence of improvement and acknowledge this case. <bold>KbPIB</bold> performs slightly worse than the baseline NF on three domains. In future, we can attempt different classifiers on privileged features and normal features.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>AUC ROC.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>NF</bold></th>
<th valign="top" align="center"><bold>KbPIB</bold></th>
<th valign="top" align="center"><bold>JPIB</bold></th>
<th valign="top" align="center"><bold>SVM&#x0002B;</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Heart</td>
<td valign="top" align="center">0.792</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.810</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.798</td>
<td valign="top" align="center">0.746</td>
</tr> <tr>
<td valign="top" align="left">Car</td>
<td valign="top" align="center">0.845</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.846</bold></td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.846</bold></td>
<td valign="top" align="center">0.841</td>
</tr> <tr>
<td valign="top" align="left">Spam</td>
<td valign="top" align="center">0.961</td>
<td valign="top" align="center">0.961</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.962</bold></td>
<td valign="top" align="center">0.934</td>
</tr> <tr>
<td valign="top" align="left">N2b_a</td>
<td valign="top" align="center">0.658</td>
<td valign="top" align="center">0.656</td>
<td valign="top" align="center" style="color:#5353ff">0.684</td>
<td valign="top" align="center"><bold>0.690</bold></td>
</tr> <tr>
<td valign="top" align="left">N2b_b</td>
<td valign="top" align="center">0.643</td>
<td valign="top" align="center" style="color:#5353ff">0.652</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.655</bold></td>
<td valign="top" align="center">0.641</td>
</tr> <tr>
<td valign="top" align="left">NS</td>
<td valign="top" align="center">0.989</td>
<td valign="top" align="center">0.989</td>
<td valign="top" align="center">0.989</td>
<td valign="top" align="center">0.5</td>
</tr> <tr>
<td valign="top" align="left">Rare</td>
<td valign="top" align="center">0.531</td>
<td valign="top" align="center" style="color:#5353ff">0.614</td>
<td valign="top" align="center" style="color:#5353ff">0.560</td>
<td valign="top" align="center"><bold>0.667</bold></td>
</tr> <tr>
<td valign="top" align="left">Adult</td>
<td valign="top" align="center">0.714</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.725</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.719</td>
<td valign="top" align="center">&#x02013;</td>
</tr> <tr>
<td valign="top" align="left">Diab.</td>
<td valign="top" align="center">0.562</td>
<td valign="top" align="center">0.561</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.566</bold></td>
<td valign="top" align="center">&#x02013;</td>
</tr> <tr>
<td valign="top" align="left">Dutch</td>
<td valign="top" align="center">0.744</td>
<td valign="top" align="center" style="color:#5353ff">0.763</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.764</bold></td>
<td valign="top" align="center">&#x02013;</td>
</tr> <tr>
<td valign="top" align="left">Bank</td>
<td valign="top" align="center">0.681</td>
<td valign="top" align="center" style="color:#5353ff">0.696</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.714</bold></td>
<td valign="top" align="center">&#x02013;</td>
</tr> <tr>
<td valign="top" align="left">Credit</td>
<td valign="top" align="center">0.701</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.703</bold></td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.703</bold></td>
<td valign="top" align="center">&#x02013;</td>
</tr> <tr>
<td valign="top" align="left">COMP.</td>
<td valign="top" align="center">0.618</td>
<td valign="top" align="center" style="color:#5353ff">0.627</td>
<td valign="top" align="center" style="color:#5353ff">0.643</td>
<td valign="top" align="center"><bold>0.698</bold></td>
</tr> <tr>
<td valign="top" align="left">C. V.</td>
<td valign="top" align="center">0.567</td>
<td valign="top" align="center" style="color:#5353ff">0.596</td>
<td valign="top" align="center" style="color:#5353ff">0.609</td>
<td valign="top" align="center"><bold>0.703</bold></td>
</tr> <tr>
<td valign="top" align="left">Comm.</td>
<td valign="top" align="center">0.893</td>
<td valign="top" align="center">0.883</td>
<td valign="top" align="center" style="color:#5353ff">0.899</td>
<td valign="top" align="center"><bold>0.919</bold></td>
</tr> <tr>
<td valign="top" align="left">St. M.</td>
<td valign="top" align="center">0.959</td>
<td valign="top" align="center" style="color:#5353ff">0.974</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.975</bold></td>
<td valign="top" align="center">0.959</td>
</tr> <tr>
<td valign="top" align="left">St. P.</td>
<td valign="top" align="center">0.908</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.921</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.914</td>
<td valign="top" align="center">0.914</td>
</tr> <tr>
<td valign="top" align="left">OUL.</td>
<td valign="top" align="center">0.523</td>
<td valign="top" align="center" style="color:#5353ff">0.532</td>
<td valign="top" align="center">0.523</td>
<td valign="top" align="center"><bold>0.534</bold></td>
</tr> <tr>
<td valign="top" align="left">KDD</td>
<td valign="top" align="center">0.889</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.890</bold></td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.890</bold></td>
<td valign="top" align="center">&#x02013;</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p><bold>KbPIB</bold> and <bold>JPIB</bold> outperform the baseline <bold>NF</bold> in nearly all the datasets. Results with standard deviation in <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>. &#x0201C;&#x02013;&#x0201D; indicates out-of-memory error. Bold values are the best scores across different methods.</p>
</table-wrap-foot>
</table-wrap>
<p>We also evaluate the approaches based on precision and recall in <xref ref-type="table" rid="T3">Table 3</xref> due to class imbalance. In 6 out of 19 datasets, our approaches yield both higher precision and recall. Our approaches achieve higher precision and higher recall in 12 datasets. Collectively, our approaches that incorporate privileged information are able to achieve better performance across several metrics (<bold>Q1</bold>).</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Precision and recall in first and second rows, respectively.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>NF</bold></th>
<th valign="top" align="center"><bold>KbPIB</bold></th>
<th valign="top" align="center"><bold>JPIB</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="2">Heart</td>
<td valign="top" align="center">0.682 &#x000B1; 0.0809</td>
<td valign="top" align="center"><inline-formula><mml:math id="M60"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>714</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1;0.0819</td>
<td valign="top" align="center"><inline-formula><mml:math id="M61"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>761</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0774</td>
</tr>
 <tr>
<td valign="top" align="center">0.786 &#x000B1; 0.1269</td>
<td valign="top" align="center"><inline-formula><mml:math id="M62"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>816</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1;0.1117</td>
<td valign="top" align="center">0.707 &#x000B1; 0.1300</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">Car</td>
<td valign="top" align="center">0.588 &#x000B1; 0.0389</td>
<td valign="top" align="center"><inline-formula><mml:math id="M63"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>592</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1;0.0384</td>
<td valign="top" align="center">0.581 &#x000B1; 0.0342</td>
</tr>
 <tr>
<td valign="top" align="center">0.908 &#x000B1; 0.0706</td>
<td valign="top" align="center">0.898 &#x000B1;0.0800</td>
<td valign="top" align="center"><bold>0.923</bold> &#x000B1; 0.0758</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">Spam</td>
<td valign="top" align="center">0.858 &#x000B1; 0.0200</td>
<td valign="top" align="center"><inline-formula><mml:math id="M64"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>873</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1;0.0218</td>
<td valign="top" align="center"><inline-formula><mml:math id="M65"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>859</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0248</td>
</tr>
 <tr>
<td valign="top" align="center">0.883 &#x000B1; 0.0251</td>
<td valign="top" align="center">0.868 &#x000B1;0.0270</td>
<td valign="top" align="center"><inline-formula><mml:math id="M66"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>884</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0268</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">N2b_a</td>
<td valign="top" align="center"><bold>0.101</bold> &#x000B1; 0.0611</td>
<td valign="top" align="center">0.078 &#x000B1;0.0476</td>
<td valign="top" align="center">0.093 &#x000B1; 0.0645</td>
</tr>
 <tr>
<td valign="top" align="center">0.553 &#x000B1; 0.3914</td>
<td valign="top" align="center"><inline-formula><mml:math id="M67"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>639</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.3821</td>
<td valign="top" align="center"><inline-formula><mml:math id="M68"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>597</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.4258</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">N2b_b</td>
<td valign="top" align="center"><bold>0.081</bold> &#x000B1; 0.0548</td>
<td valign="top" align="center">0.065 &#x000B1; 0.0503</td>
<td valign="top" align="center">0.064 &#x000B1; 0.0379</td>
</tr>
 <tr>
<td valign="top" align="center">0.572 &#x000B1; 0.3901</td>
<td valign="top" align="center"><inline-formula><mml:math id="M69"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>628</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.4250</td>
<td valign="top" align="center"><inline-formula><mml:math id="M70"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>796</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.3610</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">NS</td>
<td valign="top" align="center">0.960 &#x000B1; 0.0894</td>
<td valign="top" align="center">0.960 &#x000B1; 0.0894</td>
<td valign="top" align="center">0.960 &#x000B1; 0.0894</td>
</tr>
 <tr>
<td valign="top" align="center">0.978 &#x000B1; 0.0497</td>
<td valign="top" align="center">0.978 &#x000B1; 0.0497</td>
<td valign="top" align="center">0.978 &#x000B1; 0.0497</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">Rare</td>
<td valign="top" align="center">0.286 &#x000B1; 0.0651</td>
<td valign="top" align="center"><inline-formula><mml:math id="M71"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>340</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0885</td>
<td valign="top" align="center"><inline-formula><mml:math id="M72"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>324</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.1226</td>
</tr>
 <tr>
<td valign="top" align="center"><bold>0.879</bold> &#x000B1; 0.2174</td>
<td valign="top" align="center">0.661 &#x000B1; 0.1912</td>
<td valign="top" align="center">0.616 &#x000B1; 0.2556</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">Adult</td>
<td valign="top" align="center">0.447 &#x000B1; 0.0225</td>
<td valign="top" align="center">0.418 &#x000B1; 0.0565</td>
<td valign="top" align="center"><inline-formula><mml:math id="M73"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>452</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0371</td>
</tr>
 <tr>
<td valign="top" align="center">0.631 &#x000B1; 0.0229</td>
<td valign="top" align="center"><inline-formula><mml:math id="M74"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>704</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1;0.1212</td>
<td valign="top" align="center">0.612 &#x000B1; 0.0442</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">Diab.</td>
<td valign="top" align="center">0.243 &#x000B1; 0.0029</td>
<td valign="top" align="center"><inline-formula><mml:math id="M75"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>245</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1;0.0074</td>
<td valign="top" align="center"><inline-formula><mml:math id="M76"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>247</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0081</td>
</tr>
 <tr>
<td valign="top" align="center"><bold>0.972</bold> &#x000B1; 0.0611</td>
<td valign="top" align="center">0.946 &#x000B1; 0.1307</td>
<td valign="top" align="center">0.943 &#x000B1; 0.0864</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">Dutch</td>
<td valign="top" align="center"><bold>0.835</bold> &#x000B1; 0.0656</td>
<td valign="top" align="center">0.736 &#x000B1; 0.1237</td>
<td valign="top" align="center">0.770 &#x000B1; 0.1173</td>
</tr>
 <tr>
<td valign="top" align="center">0.572 &#x000B1; 0.0662</td>
<td valign="top" align="center"><inline-formula><mml:math id="M77"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>682</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.1439</td>
<td valign="top" align="center"><inline-formula><mml:math id="M78"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>663</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.1423</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">Bank</td>
<td valign="top" align="center">0.308 &#x000B1; 0.0300</td>
<td valign="top" align="center">0.304 &#x000B1; 0.0292</td>
<td valign="top" align="center"><inline-formula><mml:math id="M79"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>312</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0483</td>
</tr>
 <tr>
<td valign="top" align="center">0.495 &#x000B1; 0.1720</td>
<td valign="top" align="center"><inline-formula><mml:math id="M80"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>572</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.1140</td>
<td valign="top" align="center"><inline-formula><mml:math id="M81"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>522</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.1141</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">Credit</td>
<td valign="top" align="center">0.439 &#x000B1; 0.0731</td>
<td valign="top" align="center"><inline-formula><mml:math id="M82"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>487</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.1071</td>
<td valign="top" align="center"><inline-formula><mml:math id="M83"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>463</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0852</td>
</tr>
 <tr>
<td valign="top" align="center"><bold>0.599</bold> &#x000B1; 0.0951</td>
<td valign="top" align="center">0.539 &#x000B1; 0.1207</td>
<td valign="top" align="center">0.557 &#x000B1; 0.1323</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">COMP.</td>
<td valign="top" align="center">0.520 &#x000B1; 0.0476</td>
<td valign="top" align="center"><inline-formula><mml:math id="M84"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>555</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0798</td>
<td valign="top" align="center"><inline-formula><mml:math id="M85"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>559</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0698</td>
</tr>
 <tr>
<td valign="top" align="center"><bold>0.825</bold> &#x000B1; 0.1840</td>
<td valign="top" align="center">0.666 &#x000B1; 0.2403</td>
<td valign="top" align="center">0.676 &#x000B1; 0.2124</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">C. V.</td>
<td valign="top" align="center">0.271 &#x000B1; 0.0426</td>
<td valign="top" align="center"><inline-formula><mml:math id="M86"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>309</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0426</td>
<td valign="top" align="center">0.228 &#x000B1; 0.0753</td>
</tr>
 <tr>
<td valign="top" align="center">0.340 &#x000B1; 0.1536</td>
<td valign="top" align="center"><inline-formula><mml:math id="M87"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>434</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.1490</td>
<td valign="top" align="center"><inline-formula><mml:math id="M88"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>584</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.3006</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">Comm.</td>
<td valign="top" align="center"><bold>0.505</bold> &#x000B1; 0.1356</td>
<td valign="top" align="center">0.447 &#x000B1; 0.0957</td>
<td valign="top" align="center">0.485 &#x000B1;0.1200</td>
</tr>
 <tr>
<td valign="top" align="center">0.411 &#x000B1; 0.1249</td>
<td valign="top" align="center"><inline-formula><mml:math id="M89"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>515</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.1398</td>
<td valign="top" align="center"><inline-formula><mml:math id="M90"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>485</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.1805</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">St. M.</td>
<td valign="top" align="center">0.901 &#x000B1; 0.0368</td>
<td valign="top" align="center"><inline-formula><mml:math id="M91"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>989</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0486</td>
<td valign="top" align="center"><inline-formula><mml:math id="M92"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>947</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1;0.0519</td>
</tr>
 <tr>
<td valign="top" align="center"><bold>0.951</bold> &#x000B1; 0.0364</td>
<td valign="top" align="center">0.936 &#x000B1; 0.0435</td>
<td valign="top" align="center">0.925 &#x000B1; 0.0580</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">St. P.</td>
<td valign="top" align="center"><bold>0.952</bold> &#x000B1; 0.0229</td>
<td valign="top" align="center">0.938 &#x000B1; 0.0257</td>
<td valign="top" align="center">0.929 &#x000B1; 0.0337</td>
</tr>
 <tr>
<td valign="top" align="center">0.914 &#x000B1; 0.0429</td>
<td valign="top" align="center"><inline-formula><mml:math id="M93"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>969</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0272</td>
<td valign="top" align="center"><inline-formula><mml:math id="M94"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>958</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula>&#x000B1; 0.0332</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">OUL.</td>
<td valign="top" align="center">0.685 &#x000B1; 0.0114</td>
<td valign="top" align="center"><inline-formula><mml:math id="M95"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>691</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0146</td>
<td valign="top" align="center"><inline-formula><mml:math id="M96"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>688</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1;0.0125</td>
</tr>
 <tr>
<td valign="top" align="center"><bold>0.787</bold> &#x000B1; 0.0773</td>
<td valign="top" align="center">0.759 &#x000B1; 0.1103</td>
<td valign="top" align="center">0.691 &#x000B1; 0.1659</td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">KDD</td>
<td valign="top" align="center"><bold>0.391</bold> &#x000B1; 0.0176</td>
<td valign="top" align="center">0.386 &#x000B1; 0.0127</td>
<td valign="top" align="center"><bold>0.391</bold> &#x000B1; 0.0175</td>
</tr>
 <tr>
<td valign="top" align="center">0.601 &#x000B1; 0.0245</td>
<td valign="top" align="center"><inline-formula><mml:math id="M97"><mml:mrow><mml:mstyle mathcolor="#0000ff"><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>603</mml:mn></mml:mstyle></mml:mrow></mml:math></inline-formula> &#x000B1; 0.0213</td>
<td valign="top" align="center">0.601 &#x000B1; 0.0248</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p><bold>KbPIB</bold> and <bold>JPIB</bold> improve precision and recall in a big margin over the baseline <bold>NF</bold> across multiple datasets. Bold values are the best scores across different methods.</p>
</table-wrap-foot>
</table-wrap>
<p>Comparing AUC ROC in <xref ref-type="table" rid="T2">Table 2</xref>, <bold>JPIB</bold> outperforms <bold>KbPIB</bold> in 10 and achieves at least the same performance in 14 out of 19 datasets. Comparing precision and recall in <xref ref-type="table" rid="T3">Table 3</xref>, precision of <bold>JPIB</bold> outperforms <bold>KbPIB</bold> in 10 and achieves at least the same performance in 11 datasets; recall of <bold>JPIB</bold> outperforms <bold>KbPIB</bold> in 6 datasets and achieves at least the same performance in 7 datasets. Overall, <bold>JPIB</bold> outperforms <bold>KbPIB</bold>, suggesting that updating the privileged model with the classifier improves gradient boosting with sensitive information (<bold>Q2</bold>).</p>
<p>Intuitively, we expect the gains from our approach to be relative to the quality of the privileged information. When privileged information is highly discriminative, we expect greater gains from our approach and vice versa. For the standard benchmark and medical datasets (ref. <xref ref-type="table" rid="T2">Table 2</xref>), there is a correlation between the quality of the privileged information and the performance. We compare the performance only using the privileged features with <bold>KbPIB</bold> and <bold>JPIB</bold>, respectively. The Pearson correlation values of the AUC ROC are 0.237 (<bold>KbPIB</bold>) and 0.306 (<bold>JPIB</bold>). This helps explain the reason that <bold>JPIB</bold> outperforms <bold>KbPIB</bold> overall (<bold>Q2</bold>).</p>
<sec>
<title>4.3.1. Prior framework for privileged information</title>
<p>To compare with previous study of using privileged information with SVM, we run SVM&#x0002B; on our data splits and include results, as shown in <xref ref-type="table" rid="T2">Table 2</xref>. The major drawback of the previous study with SVM is that it lacks interpretability and cannot handle large datasets well. SVM&#x0002B; runs slowly on large datasets (&#x0007E;20 k instances) and fails to train on datasets with &#x0007E;30&#x0002B;k instances (6 datasets) due to out-of-memory error. As the performance difference between our method <bold>JPIB</bold> and SVM&#x0002B;, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, our method can outperform SVM&#x0002B; (<bold>Q1</bold>). For some domains, our method <bold>JPIB</bold> gets lower AUC ROC compared with SVM&#x0002B;. This shows that SVM is still a very competitive base classifier. Applying our methods to a more powerful base classifier is a prospective future study to further improve the performance on more domains.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Performance comparison between JPIB and SVM&#x0002B; (positive value if JPIB performs better; SVM&#x0002B; fails to train on six large datasets).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-06-1260583-g0002.tif"/>
</fig></sec>
<sec>
<title>4.3.2. Privileged information and fairness</title>
<p>We evaluate fairness on the fairness benchmark datasets (<xref ref-type="table" rid="T4">Table 4</xref>) and the real-world medical datasets. We compare against several fairness metrics: Statistical Parity (SP; Dwork et al., <xref ref-type="bibr" rid="B14">2012</xref>), Equalized Odds (EO; Hardt et al., <xref ref-type="bibr" rid="B19">2016</xref>), and Absolute Between-ROC Area (ABROCA; Gardner et al., <xref ref-type="bibr" rid="B17">2019</xref>). SP measures the bias of predicting positive for different groups. We use SP to measure the overall fairness in predictive accuracy of our methods. EO measures the bias of predicting positive between different groups conditioned on the label. We take EO to further examine the fairness in predictive accuracy of our methods, specifically given different labels. ABROCA measures the divergence of ROC curves between different groups. ABROCA is adopted to quantify the fairness of our methods over all possible thresholds.</p>
<disp-formula id="E12"><mml:math id="M49"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mi>S</mml:mi><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mo>|</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x00177;</mml:mi><mml:mo>=</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>|</mml:mo><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x00177;</mml:mi><mml:mo>=</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>|</mml:mo><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>E</mml:mi><mml:mi>O</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mo>,</mml:mo><mml:mo>-</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:mo>|</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x00177;</mml:mi><mml:mo>=</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>|</mml:mo><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x00177;</mml:mi><mml:mo>=</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>|</mml:mo><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>B</mml:mi><mml:mi>R</mml:mi><mml:mi>O</mml:mi><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:mo>|</mml:mo><mml:mtext>RO</mml:mtext><mml:msub><mml:mrow><mml:mtext>C</mml:mtext></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mtext>RO</mml:mtext><mml:msub><mml:mrow><mml:mtext>C</mml:mtext></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In addition to the previous baseline, we also compare against <bold>All</bold>, which learns a model that contains (imputed) privileged and classifier features. However, at test time, it estimates the privileged features based on the most common training value. Blue denotes when our approach outperforms <bold>All</bold> and bold denotes the best performance. As shown in <xref ref-type="table" rid="T4">Table 4</xref>, our approach achieves better fairness metrics than <bold>All</bold> in 10 (EO), 10 (SP), and 10 (ABROCA) datsests. Our approaches also perform at least as well as <bold>NF</bold> in 9 (EO), 9 (SP), and 13 (ABROCA) datasets. When considering the privacy or fairness of the resulting predictions, <italic>imputing the privileged information by treating them as missing</italic> has a clear <bold>negative impact on the resulting fairness (see &#x0201C;All&#x0201D; in</bold> <xref ref-type="table" rid="T4"><bold>Table 4</bold></xref><bold>)</bold>. Collectively, our approaches are able to improve performance by leveraging sensitive privileged information while maintaining fairness (<bold>Q3</bold>).</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Scores of fairness metrics (lower values are better).</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Metric</bold></th>
<th valign="top" align="center"><bold>NF</bold></th>
<th valign="top" align="center"><bold>KbPIB</bold></th>
<th valign="top" align="center"><bold>JPIB</bold></th>
<th valign="top" align="center"><bold>All</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="3">N2b_b (race)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center">0.112</td>
<td valign="top" align="center" style="color:#5353ff">0.071</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.040</bold></td>
<td valign="top" align="center">0.075</td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center">0.017</td>
<td valign="top" align="center" style="color:#5353ff">0.020</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.013</bold></td>
<td valign="top" align="center">0.023</td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.115</td>
<td valign="top" align="center" style="color:#5353ff">0.114</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.098</bold></td>
<td valign="top" align="center">0.129</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Rare (mar.)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center"><bold>0.170</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.337</td>
<td valign="top" align="center" style="color:#5353ff">0.360</td>
<td valign="top" align="center">0.526</td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center"><bold>0.067</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.126</td>
<td valign="top" align="center" style="color:#5353ff">0.130</td>
<td valign="top" align="center">0.155</td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.249</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.203</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.246</td>
<td valign="top" align="center">0.268</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Adult (sex)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center">0.371</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.314</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.346</td>
<td valign="top" align="center">0.415</td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center"><bold>0.049</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.071</td>
<td valign="top" align="center" style="color:#5353ff">0.070</td>
<td valign="top" align="center">0.081</td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.154</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.130</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.143</td>
<td valign="top" align="center">0.189</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Diab. (sex)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center">0.009</td>
<td valign="top" align="center">0.014</td>
<td valign="top" align="center">0.016</td>
<td valign="top" align="center"><bold>0.007</bold></td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center">0.005</td>
<td valign="top" align="center">0.006</td>
<td valign="top" align="center">0.007</td>
<td valign="top" align="center"><bold>0.003</bold></td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.021</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.019</bold></td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.019</bold></td>
<td valign="top" align="center">0.020</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Dutch (sex)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center">0.131</td>
<td valign="top" align="center">0.122</td>
<td valign="top" align="center">0.129</td>
<td valign="top" align="center"><bold>0.110</bold></td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center">0.094</td>
<td valign="top" align="center">0.090</td>
<td valign="top" align="center">0.087</td>
<td valign="top" align="center"><bold>0.066</bold></td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.075</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.058</bold></td>
<td valign="top" align="center">0.067</td>
<td valign="top" align="center">0.065</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Bank (age)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center">0.228</td>
<td valign="top" align="center">0.224</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.178</bold></td>
<td valign="top" align="center">0.211</td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center">0.209</td>
<td valign="top" align="center">0.221</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.183</bold></td>
<td valign="top" align="center">0.193</td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.098</td>
<td valign="top" align="center" style="color:#5353ff">0.094</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.088</bold></td>
<td valign="top" align="center">0.099</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Credit (mar.)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center">0.047</td>
<td valign="top" align="center">0.047</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.037</bold></td>
<td valign="top" align="center">0.047</td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center">0.015</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.013</bold></td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.013</bold></td>
<td valign="top" align="center">0.015</td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.029</td>
<td valign="top" align="center">0.030</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.025</bold></td>
<td valign="top" align="center">0.029</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">COMP. (race)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center">0.247</td>
<td valign="top" align="center" style="color:#5353ff">0.239</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.215</bold></td>
<td valign="top" align="center">0.334</td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center">0.146</td>
<td valign="top" align="center" style="color:#5353ff">0.141</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.135</bold></td>
<td valign="top" align="center">0.200</td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.071</td>
<td valign="top" align="center">0.059</td>
<td valign="top" align="center">0.061</td>
<td valign="top" align="center"><bold>0.037</bold></td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">C. V. (race)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center"><bold>0.190</bold></td>
<td valign="top" align="center">0.254</td>
<td valign="top" align="center" style="color:#5353ff">0.195</td>
<td valign="top" align="center">0.246</td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center"><bold>0.089</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.124</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.089</bold></td>
<td valign="top" align="center">0.162</td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.089</td>
<td valign="top" align="center">0.089</td>
<td valign="top" align="center">0.084</td>
<td valign="top" align="center"><bold>0.059</bold></td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">St. M. (sex)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center">0.210</td>
<td valign="top" align="center" style="color:#5353ff">0.187</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.184</bold></td>
<td valign="top" align="center">0.199</td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center">0.094</td>
<td valign="top" align="center">0.096</td>
<td valign="top" align="center">0.097</td>
<td valign="top" align="center"><bold>0.089</bold></td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.048</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.038</bold></td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.038</bold></td>
<td valign="top" align="center">0.052</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">St. P. (sex)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center"><bold>0.320</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.335</td>
<td valign="top" align="center" style="color:#5353ff">0.364</td>
<td valign="top" align="center">0.379</td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center">0.070</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.069</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.081</td>
<td valign="top" align="center">0.088</td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.140</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.124</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.138</td>
<td valign="top" align="center">0.152</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">OUL. (sex)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center">0.123</td>
<td valign="top" align="center">0.096</td>
<td valign="top" align="center">0.133</td>
<td valign="top" align="center"><bold>0.082</bold></td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center">0.056</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.039</bold></td>
<td valign="top" align="center">0.061</td>
<td valign="top" align="center">0.041</td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.022</td>
<td valign="top" align="center">0.022</td>
<td valign="top" align="center">0.024</td>
<td valign="top" align="center"><bold>0.021</bold></td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">KDD (race)</td>
<td valign="top" align="left">EO</td>
<td valign="top" align="center">0.100</td>
<td valign="top" align="center" style="color:#5353ff">0.101</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.099</bold></td>
<td valign="top" align="center">0.106</td>
</tr>
 <tr>
<td valign="top" align="left">SP</td>
<td valign="top" align="center"><bold>0.054</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.055</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.054</bold></td>
<td valign="top" align="center">0.060</td>
</tr>
 <tr>
<td valign="top" align="left">ABR.</td>
<td valign="top" align="center">0.033</td>
<td valign="top" align="center" style="color:#5353ff"><bold>0.032</bold></td>
<td valign="top" align="center" style="color:#5353ff">0.033</td>
<td valign="top" align="center">0.035</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p><bold>KbPIB</bold> and <bold>JPIB</bold> achieve significantly better fairness scores than the baseline <bold>All</bold> (sensitive features are imputed for test) and suppress the baseline <bold>NF</bold> over different metrics across multiple datasets. Bold values are the best scores across different methods.</p>
</table-wrap-foot>
</table-wrap>
<p>To further verify the fairness benefit of our approach, we compare with MFC (Zafar et al., <xref ref-type="bibr" rid="B52">2017</xref>). MFC learns fair classifiers by leveraging measurement of decision boundary (un)fairness, gaining fine-grained control on fairness with small cost of accuracy. As compared with MFC, our methods improve the prediction accuracy over the boosting baseline, and we would like to confirm that our methods enhance fairness. We apply MFC to our data splits and generate fairness scores on the same datasets of <xref ref-type="table" rid="T4">Table 4</xref>. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the difference of scores of three fairness metrics between our approach <bold>JPIB</bold> and the baseline MFC on each dataset. We can observe that our approach <bold>JPIB</bold> achieves comparable fairness scores to MFC.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Fairness comparison between JPIB and MFC (negative value if JPIB is fairer).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-06-1260583-g0003.tif"/>
</fig></sec></sec></sec>
<sec sec-type="conclusions" id="s5">
<title>5. Conclusion</title>
<p>We considered the problem of learning with privileged and sensitive information using gradient boosting and proposed two algorithms that learned using these information. The extensive experiments in standard, medical, and fairness datasets demonstrated the ability of our algorithms to learn robust yet fair models. More extensive evaluation on large data sets, integration of other forms of domain knowledge into our framework, understanding the relationship with other fairness models, and considering more expressive models such as deep networks remain interesting future directions.</p></sec>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>, further inquiries can be directed to the corresponding author.</p></sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>The aim of our study is to use the sensitive features as privileged ones to avoid any discriminative social bias in our study. While we do not foresee many ethical issues with our study, it is conceivable that some sensitive features might be grouped under normal feature set. The risk for ethical issues when the privileged information is identified and flagged appropriately is low. The identification of sensitive features is the most important task and could potentially affect the results of the deployment of the algorithm. The code will be released publicly and maintained by the authors in GitHub.</p></sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>SY: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing&#x02014;original draft, Writing&#x02014;review &#x00026; editing. PO: Investigation, Methodology, Writing&#x02014;original draft, Writing&#x02014;review &#x00026; editing. RP: Conceptualization, Investigation, Writing&#x02014;original draft. KK: Conceptualization, Investigation, Writing&#x02014;original draft. SN: Conceptualization, Funding acquisition, Investigation, Methodology, Project administration, Supervision, Writing&#x02014;original draft, Writing&#x02014;review &#x00026; editing.</p></sec>
</body>
<back>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. SY and SN gratefully acknowledge AFOSR Minerva award FA9550-19-1-039. KK gratefully acknowledges the Hessian Ministry of Higher Education, Research, Science and the Arts (HMWK) cluster project The Third Wave of AI as well as the project safeFBDC&#x02014;Financial Big Data Cluster (FKZ: 01MK21002K), funded by the German Federal Ministry for Economics Affairs and Energy as part of the GAIA-x initiative.</p>
</sec>
<ack><p>The authors acknowledge the support of members of STARLING lab for the discussions. The authors also thank the reviewers for their insightful comments and in significantly improving the study.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>RP was employed by Amazon. The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Author disclaimer</title>
<p>Any opinions, findings, and conclusion or recommendations expressed in this material are those of the authors and do not necessarily reflect the view of the AFOSR or the US government.</p>
</sec>
<sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2023.1260583/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frai.2023.1260583/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.PDF" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup>We use <inline-formula><mml:math id="M24"><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> to denote the probability mass of being positive given the classifier features.</p></fn>
<fn id="fn0002"><p><sup>2</sup>More details of derivation in <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>.</p></fn>
<fn id="fn0003"><p><sup>3</sup><ext-link ext-link-type="uri" xlink:href="https://github.com/starling-lab/PI_GBM">https://github.com/starling-lab/PI_GBM</ext-link></p></fn>
<fn id="fn0004"><p><sup>4</sup><ext-link ext-link-type="uri" xlink:href="https://www.lalpathlabs.com/">https://www.lalpathlabs.com/</ext-link></p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Altendorf</surname> <given-names>E.</given-names></name> <name><surname>Restificar</surname> <given-names>A.</given-names></name> <name><surname>Dietterich</surname> <given-names>T.</given-names></name></person-group> (<year>2005</year>). <article-title>&#x0201C;Learning from sparse data by exploiting monotonicity constraints,&#x0201D;</article-title> in <source>UAI&#x00027;05: Proceedings of the Twenty-First Conference on Uncertainty in Artificial Intelligence</source> (<publisher-loc>Edinburgh</publisher-loc>: <publisher-name>AUAI Press</publisher-name>), <fpage>18</fpage>&#x02013;<lpage>26</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Angwin</surname> <given-names>J.</given-names></name> <name><surname>Larson</surname> <given-names>J.</given-names></name> <name><surname>Mattu</surname> <given-names>S.</given-names></name> <name><surname>Kirchner</surname> <given-names>L.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Machine bias,&#x0201D;</article-title> in <source>Ethics of Data and Analytics</source> (<publisher-loc>Auerbach Publications</publisher-loc>), <fpage>254</fpage>&#x02013;<lpage>264</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Boutilier</surname> <given-names>C.</given-names></name></person-group> (<year>2002</year>). <article-title>&#x0201C;A POMDP formulation of preference elicitation problems,&#x0201D;</article-title> in <source>Proceedings of the Eighteenth National Conference on Artificial Intelligence and Fourteenth Conference on Innovative Applications of Artificial Intelligence</source>, eds <person-group person-group-type="editor"><name><surname>Dechter</surname> <given-names>R.</given-names></name> <name><surname>Kearns</surname> <given-names>M. J.</given-names></name> <name><surname>Sutton</surname> <given-names>R. S.</given-names></name></person-group> (<publisher-loc>Edmonton, AB</publisher-loc>: <publisher-name>AAAI Press; The MIT Press</publisher-name>), <fpage>239</fpage>&#x02013;<lpage>246</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.aaai.org/Library/AAAI/2002/aaai02-037.php">http://www.aaai.org/Library/AAAI/2002/aaai02-037.php</ext-link></citation>
</ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bu</surname> <given-names>S.</given-names></name> <name><surname>Cho</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Integrating deep learning with first-order logic programmed constraints for zero-day phishing attack detection,&#x0201D;</article-title> in <source>IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP</source> (<publisher-loc>Toronto, ON</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2685</fpage>&#x02013;<lpage>2689</lpage>. <pub-id pub-id-type="doi">10.1109/ICASSP39728.2021.9414850</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Lyu</surname> <given-names>S.</given-names></name></person-group> (<year>2012</year>). <article-title>&#x0201C;Boosting with side information,&#x0201D;</article-title> in <source>11th Asian Conference on Computer Vision</source>, eds <person-group person-group-type="editor"><name><surname>Lee</surname> <given-names>K. M.</given-names></name> <name><surname>Matsushita</surname> <given-names>Y.</given-names></name> <name><surname>Rehg</surname> <given-names>J. M.</given-names></name> <name><surname>Hu</surname> <given-names>Z.</given-names></name></person-group> (<publisher-loc>Daejeon</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>563</fpage>&#x02013;<lpage>577</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-642-37331-2_43</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Choudhuri</surname> <given-names>A.</given-names></name> <name><surname>Green</surname> <given-names>M.</given-names></name> <name><surname>Jain</surname> <given-names>A.</given-names></name> <name><surname>Kaptchuk</surname> <given-names>G.</given-names></name> <name><surname>Miers</surname> <given-names>I.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Fairness in an unfair world: fair multiparty computation from public bulletin boards,&#x0201D;</article-title> in <source>Proceedings of the 2017 ACM SIGSAC Conference on Computer and Communications Security</source>, eds <person-group person-group-type="editor"><name><surname>Thuraisingham</surname> <given-names>B.</given-names></name> <name><surname>Evans</surname> <given-names>D.</given-names></name> <name><surname>Malkin</surname> <given-names>T.</given-names></name> <name><surname>Xu</surname> <given-names>D.</given-names></name></person-group> (<publisher-loc>Dallas, TX</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>719</fpage>&#x02013;<lpage>728</lpage>. <pub-id pub-id-type="doi">10.1145/3133956.3134092</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chouldechova</surname> <given-names>A.</given-names></name> <name><surname>Benavides-Prado</surname> <given-names>D.</given-names></name> <name><surname>Fialko</surname> <given-names>O.</given-names></name> <name><surname>Vaithianathan</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;A case study of algorithm-assisted decision making in child maltreatment hotline screening decisions,&#x0201D;</article-title> in <source>Conference on Fairness, Accountability and Transparency, FAT 2018</source>, eds <person-group person-group-type="editor"><name><surname>Friedler</surname> <given-names>S. A.</given-names></name> <name><surname>Wilson</surname> <given-names>C.</given-names></name></person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>134</fpage>&#x02013;<lpage>148</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cortez</surname> <given-names>P.</given-names></name> <name><surname>Silva</surname> <given-names>A.</given-names></name></person-group> (<year>2008</year>). <source>Using Data Mining to Predict Secondary School Student Performance</source>. EUROSIS-ETI.</citation>
</ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Das</surname> <given-names>M.</given-names></name> <name><surname>Dhami</surname> <given-names>D.</given-names></name> <name><surname>Yu</surname> <given-names>Y.</given-names></name> <name><surname>Kunapuli</surname> <given-names>G.</given-names></name> <name><surname>Natarajan</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Human-guided learning of column networks: knowledge injection for relational deep learning,&#x0201D;</article-title> in <source>CODS-COMAD &#x00027;21: Proceedings of the 3rd ACM India Joint International Conference on Data Science &#x00026; Management of Data (8th ACM IKDD CODS &#x00026; 26th COMAD)</source>, eds <person-group person-group-type="editor"><name><surname>Haritsa</surname> <given-names>J. R.</given-names></name> <name><surname>Roy</surname> <given-names>S.</given-names></name> <name><surname>Gupta</surname> <given-names>M.</given-names></name> <name><surname>Mehrotra</surname> <given-names>S.</given-names></name> <name><surname>Srinivasan</surname> <given-names>B. V.</given-names></name> <name><surname>Simmhan</surname> <given-names>Y.</given-names></name></person-group> (<publisher-loc>ACM</publisher-loc>: <publisher-name>Bengaluru</publisher-name>), <fpage>110</fpage>&#x02013;<lpage>118</lpage>. <pub-id pub-id-type="doi">10.1145/3430984.3431018</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Dheeru</surname> <given-names>D.</given-names></name> <name><surname>Taniskidou</surname> <given-names>E.</given-names></name></person-group> (<year>2017</year>). <source>The UCI Machine Learning Repository</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://archive.ics.uci.edu">https://archive.ics.uci.edu</ext-link></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dietterich</surname> <given-names>T.</given-names></name> <name><surname>Hao</surname> <given-names>G.</given-names></name> <name><surname>Ashenfelter</surname> <given-names>A.</given-names></name></person-group> (<year>2008</year>). <article-title>Gradient tree boosting for training conditional random fields</article-title>. <source>J. Mach. Learn. Res.</source> 9.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>X.</given-names></name> <name><surname>Luo</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Q.</given-names></name> <name><surname>Cheng</surname> <given-names>Y.</given-names></name> <name><surname>Cai</surname> <given-names>G.</given-names></name> <name><surname>Munnoch</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Prior knowledge-based deep learning method for indoor object recognition and application</article-title>. <source>Syst. Sci. Control</source> <volume>6</volume>, <fpage>249</fpage>&#x02013;<lpage>251</lpage>. <pub-id pub-id-type="doi">10.1080/21642583.2018.1482477</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Drummond</surname> <given-names>J.</given-names></name> <name><surname>Boutilier</surname> <given-names>C.</given-names></name></person-group> (<year>2014</year>). <article-title>&#x0201C;Preference elicitation and interview minimization in stable matchings,&#x0201D;</article-title> in <source>Proceedings of the Twenty-Eighth AAAI Conference on Artificial Intelligence</source>, eds <person-group person-group-type="editor"><name><surname>Brodley</surname> <given-names>C. E.</given-names></name> <name><surname>Stone</surname> <given-names>P.</given-names></name></person-group> (<publisher-loc>Qu&#x000E9;bec City</publisher-loc>: <publisher-name>AAAI Press</publisher-name>), <fpage>645</fpage>&#x02013;<lpage>653</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v28i1.8829</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dwork</surname> <given-names>C.</given-names></name> <name><surname>Hardt</surname> <given-names>M.</given-names></name> <name><surname>Pitassi</surname> <given-names>T.</given-names></name> <name><surname>Reingold</surname> <given-names>O.</given-names></name> <name><surname>Zemel</surname> <given-names>R.</given-names></name></person-group> (<year>2012</year>). <article-title>&#x0201C;Fairness through awareness,&#x0201D;</article-title> in <source>Innovations in Theoretical Computer Science 2012</source>, ed <person-group person-group-type="editor"><name><surname>Goldwasser</surname> <given-names>S.</given-names></name></person-group> (<publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>214</fpage>&#x02013;<lpage>226</lpage>. <pub-id pub-id-type="doi">10.1145/2090236.2090255</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friedman</surname> <given-names>J.</given-names></name></person-group> (<year>2001</year>). <article-title>Greedy function approximation: a gradient boosting machine</article-title>. <source>Ann. Stat</source>. <volume>29</volume>, <fpage>1189</fpage>&#x02013;<lpage>1232</lpage>. <pub-id pub-id-type="doi">10.1214/aos/1013203451</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Fung</surname> <given-names>G.</given-names></name> <name><surname>Mangasarian</surname> <given-names>O.</given-names></name> <name><surname>Shavlik</surname> <given-names>J.</given-names></name></person-group> (<year>2002</year>). <article-title>&#x0201C;Knowledge-Based support vector machine classifiers,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 15 (NIPS 2002)</source>, eds S. Becker, S. Thrun, and K. Obermayer (<publisher-loc>Vancouver, BC</publisher-loc>: <publisher-name>MIT Press</publisher-name>), <fpage>521</fpage>&#x02013;<lpage>528</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gardner</surname> <given-names>J.</given-names></name> <name><surname>Brooks</surname> <given-names>C.</given-names></name> <name><surname>Baker</surname> <given-names>R.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Evaluating the fairness of predictive student models through slicing analysis,&#x0201D;</article-title> in <source>LAK19: Proceedings of the 9th International Conference on Learning Analytics</source> &#x00026; <italic>Knowledge</italic> (Tempe, AZ: ACM), <fpage>225</fpage>&#x02013;<lpage>234</lpage>. <pub-id pub-id-type="doi">10.1145/3303772.3303791</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haas</surname> <given-names>D.</given-names></name> <name><surname>Parker</surname> <given-names>C.</given-names></name> <name><surname>Wing</surname> <given-names>D.</given-names></name> <name><surname>Parry</surname> <given-names>S.</given-names></name> <name><surname>Grobman</surname> <given-names>W.</given-names></name> <name><surname>Mercer</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>A description of the methods of the nulliparous pregnancy outcomes study: monitoring mothers-to-be (numom2b)</article-title>. <source>Am. J. Obstet. Gynecol</source>. <volume>212</volume>, <fpage>539.e1</fpage>&#x02013;<lpage>539.e24</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajog.2015.01.019</pub-id><pub-id pub-id-type="pmid">25648779</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hardt</surname> <given-names>M.</given-names></name> <name><surname>Price</surname> <given-names>E.</given-names></name> <name><surname>Srebro</surname> <given-names>N.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Equality of opportunity in supervised learning,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 29: Annual Conference on Neural Information Processing Systems 2016</source>, eds D. D. Lee, M. Sugiyama, U. von Luxburg, I. Guyon, and R. Garnett (Barcelona), <fpage>3315</fpage>&#x02013;<lpage>3323</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hern&#x000E1;ndez-Lobato</surname> <given-names>D.</given-names></name> <name><surname>Sharmanska</surname> <given-names>V.</given-names></name> <name><surname>Kersting</surname> <given-names>K.</given-names></name> <name><surname>Lampert</surname> <given-names>C.</given-names></name> <name><surname>Quadrianto</surname> <given-names>N.</given-names></name></person-group> (<year>2014</year>). <article-title>&#x0201C;Mind the nuisance: Gaussian process classification using privileged noise,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 27: Annual Conference on Neural Information Processing Systems 2014</source>, eds Z. Ghahramani, M. Welling, C. Cortes, N. D. Lawrence, and K. Q. Weinberger (<publisher-loc>Montreal, QC</publisher-loc>), <fpage>837</fpage>&#x02013;<lpage>845</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G.</given-names></name> <name><surname>Vinyals</surname> <given-names>O.</given-names></name> <name><surname>Dean</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). <article-title>Distilling the knowledge in a neural network</article-title>. <source>arXiv:1503.02531</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1503.02531</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Joachims</surname> <given-names>T.</given-names></name></person-group> (<year>1999</year>). <article-title>&#x0201C;Transductive inference for text classification using support vector machines,&#x0201D;</article-title> in <source>Proceedings of the Sixteenth International Conference on Machine Learning (ICML 1999)</source>, eds <person-group person-group-type="editor"><name><surname>Bratko</surname> <given-names>I.</given-names></name> <name><surname>Dzeroski</surname> <given-names>S.</given-names></name></person-group> (<publisher-loc>Bled</publisher-loc>: <publisher-name>Morgan Kaufmann</publisher-name>), <fpage>200</fpage>&#x02013;<lpage>209</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kamishima</surname> <given-names>T.</given-names></name> <name><surname>Akaho</surname> <given-names>S.</given-names></name> <name><surname>Asoh</surname> <given-names>H.</given-names></name> <name><surname>Sakuma</surname> <given-names>J.</given-names></name></person-group> (<year>2012</year>). <article-title>&#x0201C;Fairness-aware classifier with prejudice remover regularizer,&#x0201D;</article-title> in <source>Machine Learning and Knowledge Discovery in Databases - European Conference, ECML PKDD 2012</source>, eds P. A. Flach, T. De Bie, and N. Cristianini (<publisher-loc>Bristol</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>35</fpage>&#x02013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-642-33486-3_3</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kilbertus</surname> <given-names>N.</given-names></name> <name><surname>Gasc&#x000F3;n</surname> <given-names>A.</given-names></name> <name><surname>Kusner</surname> <given-names>M.</given-names></name> <name><surname>Veale</surname> <given-names>M.</given-names></name> <name><surname>Gummadi</surname> <given-names>K.</given-names></name> <name><surname>Weller</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Blind justice: fairness with encrypted sensitive attributes,&#x0201D;</article-title> in <source>Proceedings of the 35th International Conference on Machine Learning, ICML 2018</source>, eds J. G. Dy and A. Krause (<publisher-loc>Stockholm</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>2635</fpage>&#x02013;<lpage>2644</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Kohavi</surname> <given-names>R.</given-names></name></person-group> (<year>1996</year>). <article-title>&#x0201C;Scaling up the accuracy of naive-Bayes classifiers: a decision-tree hybrid,&#x0201D;</article-title> in <source>Proceedings of the Second International Conference on Knowledge Discovery and Data Mining (KDD-96)</source>, eds <person-group person-group-type="editor"><name><surname>Simoudis</surname> <given-names>E.</given-names></name> <name><surname>Han</surname> <given-names>J.</given-names></name> <name><surname>Fayyad</surname> <given-names>U. M.</given-names></name></person-group> (<publisher-loc>Portland, OR</publisher-loc>: <publisher-name>AAAI Press</publisher-name>), <fpage>202</fpage>&#x02013;<lpage>207</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.aaai.org/Library/KDD/1996/kdd96-033.php">http://www.aaai.org/Library/KDD/1996/kdd96-033.php</ext-link></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kokel</surname> <given-names>H.</given-names></name> <name><surname>Odom</surname> <given-names>P.</given-names></name> <name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Natarajan</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>A unified framework for knowledge intensive gradient boosting: leveraging human experts for noisy sparse domains</article-title>. <source>Proc. AAAI Conf. Artif. Intell</source>. <volume>34</volume>, <fpage>4460</fpage>&#x02013;<lpage>4468</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v34i04.5873</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Krasanakis</surname> <given-names>E.</given-names></name> <name><surname>Spyromitros-Xioufis</surname> <given-names>E.</given-names></name> <name><surname>Papadopoulos</surname> <given-names>S.</given-names></name> <name><surname>Kompatsiaris</surname> <given-names>Y.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Adaptive sensitive reweighting to mitigate bias in fairness-aware classification,&#x0201D;</article-title> in <source>WWW &#x00027;18: Proceedings of the 2018 World Wide Web Conference</source>, eds P.-A. Champin, F. Gandon, M. Lalmas, and P. G. Ipeirotis (<publisher-loc>Lyon</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>853</fpage>&#x02013;<lpage>862</lpage>. <pub-id pub-id-type="doi">10.1145/3178876.3186133</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kunapuli</surname> <given-names>G.</given-names></name> <name><surname>Odom</surname> <given-names>P.</given-names></name> <name><surname>Shavlik</surname> <given-names>J.</given-names></name> <name><surname>Natarajan</surname> <given-names>S.</given-names></name></person-group> (<year>2013</year>). <article-title>&#x0201C;Guiding autonomous agents to better behaviors through human advice,&#x0201D;</article-title> in <source>2013 IEEE 13th International Conference on Data Mining</source>, eds H. Xiong, G. Karypis, B. Thuraisingham, D. J. Cook, and X. Wu (<publisher-loc>Dallas, TX</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>409</fpage>&#x02013;<lpage>418</lpage>. <pub-id pub-id-type="doi">10.1109/ICDM.2013.79</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuzilek</surname> <given-names>J.</given-names></name> <name><surname>Hlosta</surname> <given-names>M.</given-names></name> <name><surname>Zdrahal</surname> <given-names>Z.</given-names></name></person-group> (<year>2017</year>). <article-title>Open university learning analytics dataset</article-title>. <source>Sci. Data</source>. <volume>4</volume>, <fpage>170171</fpage>. <pub-id pub-id-type="doi">10.1038/sdata.2017.171</pub-id><pub-id pub-id-type="pmid">29182599</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lapin</surname> <given-names>M.</given-names></name> <name><surname>Hein</surname> <given-names>M.</given-names></name> <name><surname>Schiele</surname> <given-names>B.</given-names></name></person-group> (<year>2014</year>). <article-title>Learning using privileged information: SV M&#x0002B; and weighted SVM</article-title>. <source>Neural Netw</source>. <volume>53</volume>, <fpage>95</fpage>&#x02013;<lpage>108</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2014.02.002</pub-id><pub-id pub-id-type="pmid">24576747</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liang</surname> <given-names>L.</given-names></name> <name><surname>Cai</surname> <given-names>F.</given-names></name> <name><surname>Cherkassky</surname> <given-names>V.</given-names></name></person-group> (<year>2009</year>). <article-title>Predictive learning with structured (grouped) data</article-title>. <source>Neural Netw</source>. <volume>22</volume>, <fpage>766</fpage>&#x02013;<lpage>773</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2009.06.030</pub-id><pub-id pub-id-type="pmid">19596546</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Lopez-Paz</surname> <given-names>D.</given-names></name> <name><surname>Bottou</surname> <given-names>L.</given-names></name> <name><surname>Sch&#x000F6;lkopf</surname> <given-names>B.</given-names></name> <name><surname>Vapnik</surname> <given-names>V.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Unifying distillation and privileged information,&#x0201D;</article-title> in <source>4th International Conference on Learning Representations, ICLR 2016</source>, eds <person-group person-group-type="editor"><name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>LeCun</surname> <given-names>Y.</given-names></name></person-group> (San Juan, PR). Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1511.03643">http://arxiv.org/abs/1511.03643</ext-link></citation>
</ref>
<ref id="B33">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>MacLeod</surname> <given-names>H.</given-names></name> <name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Oakes</surname> <given-names>K.</given-names></name> <name><surname>Connelly</surname> <given-names>K.</given-names></name> <name><surname>Natarajan</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Identifying rare diseases from behavioural data: a machine learning approach,&#x0201D;</article-title> in <source>Proceedings of the First IEEE International Conference on Connected Health: Applications, Systems and Engineering Technologies, CHASE, 2016</source> (<publisher-loc>Washington, DC</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>130</fpage>&#x02013;<lpage>139</lpage>. <pub-id pub-id-type="doi">10.1109/CHASE.2016.7</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Maclin</surname> <given-names>R.</given-names></name> <name><surname>Shavlik</surname> <given-names>J.</given-names></name> <name><surname>Torrey</surname> <given-names>L.</given-names></name> <name><surname>Walker</surname> <given-names>T.</given-names></name> <name><surname>Wild</surname> <given-names>E.</given-names></name></person-group> (<year>2005</year>). <article-title>&#x0201C;Giving advice about preferred actions to reinforcement learners via knowledge-based kernel regression,&#x0201D;</article-title> in <source>Proceedings, the Twentieth National Conference on Artificial Intelligence and the Seventeenth Innovative Applications of Artificial Intelligence Conference</source>, eds <person-group person-group-type="editor"><name><surname>Veloso</surname> <given-names>M. M.</given-names></name> <name><surname>Kambhampati</surname> <given-names>S.</given-names></name></person-group> (Pittsburgh, PA: AAAI Press/The MIT Press), <fpage>819</fpage>&#x02013;<lpage>824</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.aaai.org/Library/AAAI/2005/aaai05-129.php">http://www.aaai.org/Library/AAAI/2005/aaai05-129.php</ext-link></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moro</surname> <given-names>S.</given-names></name> <name><surname>Cortez</surname> <given-names>P.</given-names></name> <name><surname>Rita</surname> <given-names>P.</given-names></name></person-group> (<year>2014</year>). <article-title>A data-driven approach to predict the success of bank telemarketing</article-title>. <source>Decis. Support Syst</source>. <volume>62</volume>, <fpage>22</fpage>&#x02013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1016/j.dss.2014.03.001</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Natarajan</surname> <given-names>S.</given-names></name> <name><surname>Kersting</surname> <given-names>K.</given-names></name> <name><surname>Khot</surname> <given-names>T.</given-names></name> <name><surname>Shavlik</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). <source>Boosted Statistical Relational Learners: From Benchmarks to Data-Driven Medicine.</source> Springer. <pub-id pub-id-type="doi">10.1007/978-3-319-13644-8</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Natarajan</surname> <given-names>S.</given-names></name> <name><surname>Khot</surname> <given-names>T.</given-names></name> <name><surname>Kersting</surname> <given-names>K.</given-names></name> <name><surname>Gutmann</surname> <given-names>B.</given-names></name> <name><surname>Shavlik</surname> <given-names>J.</given-names></name></person-group> (<year>2012</year>). <article-title>Gradient-based boosting for statistical relational learning: the relational dependency network case</article-title>. <source>Mach. Learn</source>. <volume>86</volume>, <fpage>25</fpage>&#x02013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1007/s10994-011-5244-9</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pang</surname> <given-names>S.</given-names></name> <name><surname>Orgun</surname> <given-names>M.</given-names></name> <name><surname>Yu</surname> <given-names>Z.</given-names></name></person-group> (<year>2018</year>). <article-title>A novel biomedical image indexing and retrieval system via deep preference learning</article-title>. <source>Comput. Methods Prog. Biomed</source>. <volume>158</volume>, <fpage>53</fpage>&#x02013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmpb.2018.02.003</pub-id><pub-id pub-id-type="pmid">29544790</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Pechyony</surname> <given-names>D.</given-names></name> <name><surname>Vapnik</surname> <given-names>V.</given-names></name></person-group> (<year>2010</year>). <article-title>&#x0201C;On the theory of learning with privileged information,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 23: 24th Annual Conference on Neural Information Processing Systems 2010</source>, eds J. D. Lafferty, C. K. I. Williams, J. Shawe-Taylor, R. S. Zemel, and A. Culotta (<publisher-loc>Vancouver, BC</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>), <fpage>1894</fpage>&#x02013;<lpage>1902</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Quadrianto</surname> <given-names>N.</given-names></name> <name><surname>Sharmanska</surname> <given-names>V.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Recycling privileged learning and distribution matching for fairness,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017</source>, eds I. Guyon, U. von Luxburg, S. Bengio, H. M. Wallach, R. Fergus, S. V. N. Vishwanathan, and R. Garnett (<publisher-loc>Long Beach, CA</publisher-loc>), <fpage>677</fpage>&#x02013;<lpage>688</lpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Settles</surname> <given-names>B.</given-names></name></person-group> (<year>2012</year>). <article-title>Active Learning</article-title>. <source>Synthesis Lectures on Artificial Intelligence and Machine Learning.</source> Morgan &#x00026; Claypool. <pub-id pub-id-type="doi">10.1007/978-3-031-01560-1</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sharmanska</surname> <given-names>V.</given-names></name> <name><surname>Quadrianto</surname> <given-names>N.</given-names></name> <name><surname>Lampert</surname> <given-names>C.</given-names></name></person-group> (<year>2013</year>). <article-title>&#x0201C;Learning to rank using privileged information,&#x0201D;</article-title> in <source>CVPR</source>. <pub-id pub-id-type="doi">10.1109/ICCV.2013.107</pub-id></citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Strack</surname> <given-names>B.</given-names></name> <name><surname>DeShazo</surname> <given-names>J. P.</given-names></name> <name><surname>Gennings</surname> <given-names>C.</given-names></name> <name><surname>Olmo</surname> <given-names>J. L.</given-names></name> <name><surname>Ventura</surname> <given-names>S.</given-names></name> <name><surname>Cios</surname> <given-names>K. J.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Impact of hba1c measurement on hospital readmission rates: analysis of 70,000 clinical database patient records</article-title>. <source>BioMed Res. Int.</source> 2014. <pub-id pub-id-type="doi">10.1155/2014/781670</pub-id><pub-id pub-id-type="pmid">24804245</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><collab>Towell G. and Shavlik, J.</collab></person-group> (<year>1994</year>). <article-title>Knowledge-based artificial neural networks</article-title>. <source>Artif. Intell.</source> <volume>70</volume>, <fpage>119</fpage>&#x02013;<lpage>165</lpage>. <pub-id pub-id-type="doi">10.1016/0004-3702(94)90105-8</pub-id></citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van der Laan</surname> <given-names>P.</given-names></name></person-group> (<year>2000</year>). <article-title>&#x0201C;The 2001 census in the Netherlands,&#x0201D;</article-title> in <source>Conference the Census of Population</source>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vapnik</surname> <given-names>V.</given-names></name> <name><surname>Vashist</surname> <given-names>A.</given-names></name></person-group> (<year>2009</year>). <article-title>A new learning paradigm: learning using privileged information</article-title>. <source>Neural Netw</source>. <volume>22</volume>, <fpage>544</fpage>&#x02013;<lpage>557</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2009.06.042</pub-id></citation>
</ref>
<ref id="B47">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Gao</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Fair classification under strict unawareness,&#x0201D;</article-title> in <source>Proceedings of the 2021 SIAM International Conference on Data Mining, SDM 2021</source>, eds <person-group person-group-type="editor"><name><surname>Demeniconi</surname> <given-names>C.</given-names></name> <name><surname>Davidson</surname> <given-names>I</given-names></name></person-group> (SIAM), <fpage>199</fpage>&#x02013;<lpage>207</lpage>. <pub-id pub-id-type="doi">10.1137/1.9781611976700.23</pub-id></citation>
</ref>
<ref id="B48">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Pan</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Integrating deep learning with logic fusion for information extraction,&#x0201D;</article-title> in <source>The Thirty-Fourth AAAI Conference on Artificial Intelligence, AAAI 2020, The Thirty-Second Innovative Applications of Artificial Intelligence Conference, IAAI 2020, The Tenth AAAI Symposium on Educational Advances in Artificial Intelligence, EAAI 2020</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>AAAI Press</publisher-name>), <fpage>9225</fpage>&#x02013;<lpage>9232</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v34i05.6460</pub-id></citation>
</ref>
<ref id="B49">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Williamson</surname> <given-names>R.</given-names></name> <name><surname>Menon</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Fairness risk measures,&#x0201D;</article-title> in <source>Proceedings of the 36th International Conference on Machine Learning, ICML 2019</source>, eds <person-group person-group-type="editor"><name><surname>Chaudhuri</surname> <given-names>K.</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R.</given-names></name></person-group> (<publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>6786</fpage>&#x02013;<lpage>6797</lpage>.</citation>
</ref>
<ref id="B50">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Khot</surname> <given-names>T.</given-names></name> <name><surname>Kersting</surname> <given-names>K.</given-names></name> <name><surname>Natarajan</surname> <given-names>S.</given-names></name></person-group> (<year>2013</year>). <article-title>&#x0201C;Knowledge intensive learning: combining qualitative constraints with causal independence for parameter learning in probabilistic models,&#x0201D;</article-title> in <source>Machine Learning and Knowledge Discovery in Databases - European Conference, ECML PKDD 2013</source>, eds <person-group person-group-type="editor"><name><surname>Blockeel</surname> <given-names>H.</given-names></name> <name><surname>Kersting</surname> <given-names>K.</given-names></name> <name><surname>Nijssen</surname> <given-names>S.</given-names></name> <name><surname>Zelezn&#x000FD;</surname> <given-names>F.</given-names></name></person-group> (<publisher-loc>Prague</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>580</fpage>&#x02013;<lpage>595</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-642-40991-2_37</pub-id></citation>
</ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yeh</surname> <given-names>I.</given-names></name> <name><surname>Lien</surname> <given-names>C.</given-names></name></person-group> (<year>2009</year>). <article-title>The comparisons of data mining techniques for the predictive accuracy of probability of default of credit card clients</article-title>. <source>Expert Syst. Appl</source>. <volume>36</volume>, <fpage>2473</fpage>&#x02013;<lpage>2480</lpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2007.12.020</pub-id></citation>
</ref>
<ref id="B52">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zafar</surname> <given-names>M.</given-names></name> <name><surname>Valera</surname> <given-names>I.</given-names></name> <name><surname>Rodriguez</surname> <given-names>M.</given-names></name> <name><surname>Gummadi</surname> <given-names>K.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Fairness constraints: mechanisms for fair classification,&#x0201D;</article-title> in <source>Proceedings of the 20th International Conference on Artificial Intelligence and Statistics, AISTATS 2017</source>, eds <person-group person-group-type="editor"><name><surname>Singh</surname> <given-names>A.</given-names></name> <name><surname>Zhu</surname> <given-names>X.</given-names></name></person-group> (<publisher-loc>Fort Lauderdale, FL</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>962</fpage>&#x02013;<lpage>970</lpage>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>&#x0017D;liobait&#x00117;</surname> <given-names>I.</given-names></name></person-group> (<year>2017</year>). <article-title>Measuring discrimination in algorithmic decision making</article-title>. <source>Data Mining Knowl. Discov</source>. <volume>31</volume>, <fpage>1060</fpage>&#x02013;<lpage>1089</lpage>. <pub-id pub-id-type="doi">10.1007/s10618-017-0506-1</pub-id></citation>
</ref>
</ref-list>
</back>
</article> 