<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurorobot.</journal-id>
<journal-title>Frontiers in Neurorobotics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurorobot.</abbrev-journal-title>
<issn pub-type="epub">1662-5218</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnbot.2024.1391247</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>The meta-learning method for the ensemble model based on situational meta-task</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Zhengchao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhou</surname> <given-names>Lianke</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2665982/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wu</surname> <given-names>Yuyang</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Nianbin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>College of Computer Science and Technology, Harbin Engineering University, Harbin</institution>, <addr-line>Heilongjiang</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Modeling and Emulation in E-Government National Engineering Laboratory, Harbin Engineering University, Harbin</institution>, <addr-line>Heilongjiang</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>School of Computer Science and Technology, Guangdong University of Technology, Guangzhou</institution>, <addr-line>Guangdong</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Xianmin Wang, Guangzhou University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Jianzong Wang, Ping An Technology Co., Ltd., China</p>
<p>Anna Vettoruzzo, Halmstad University, Sweden</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Lianke Zhou <email>zhoulianke&#x00040;hrbeu.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>26</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>18</volume>
<elocation-id>1391247</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Zhang, Zhou, Wu and Wang.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Zhang, Zhou, Wu and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>The meta-learning methods have been widely used to solve the problem of few-shot learning. Generally, meta-learners are trained on a variety of tasks and then generalized to novel tasks.</p></sec>
<sec>
<title>Methods</title>
<p>However, existing meta-learning methods do not consider the relationship between meta-tasks and novel tasks during the meta-training period, so that initial models of the meta-learner provide less useful meta-knowledge for the novel tasks. This leads to a weak generalization ability on novel tasks. Meanwhile, different initial models contain different meta-knowledge, which leads to certain differences in the learning effect of novel tasks during the meta-testing period. Therefore, this article puts forward a meta-optimization method based on situational meta-task construction and cooperation of multiple initial models. First, during the meta-training period, a method of constructing situational meta-task is proposed, and the selected candidate task sets provide more effective meta-knowledge for novel tasks. Then, during the meta-testing period, an ensemble model method based on meta-optimization is proposed to minimize the loss of inter-model cooperation in prediction, so that multiple models cooperation can realize the learning of novel tasks.</p></sec>
<sec>
<title>Results</title>
<p>The above-mentioned methods are applied to popular few-shot character datasets and image recognition datasets. Furthermore, the experiment results indicate that the proposed method achieves good effects in few-shot classification tasks.</p></sec>
<sec>
<title>Discussion</title>
<p>In future work, we will extend our methods to provide more generalized and useful meta-knowledge to the model during the meta-training period when the novel few-shot tasks are completely invisible.</p></sec></abstract>
<kwd-group>
<kwd>meta-learning</kwd>
<kwd>few-shot learning</kwd>
<kwd>situational meta-task</kwd>
<kwd>ensemble model</kwd>
<kwd>image recognition</kwd>
</kwd-group>
<contract-sponsor id="cn001">Harbin Engineering University<named-content content-type="fundref-id">10.13039/501100003471</named-content></contract-sponsor>
<counts>
<fig-count count="10"/>
<table-count count="4"/>
<equation-count count="11"/>
<ref-count count="46"/>
<page-count count="15"/>
<word-count count="8861"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Deep learning (LeCun et al., <xref ref-type="bibr" rid="B18">2015</xref>) has achieved great success and has become a practical method in many applications, such as computer vision (Yu et al., <xref ref-type="bibr" rid="B40">2022</xref>; Jiang et al., <xref ref-type="bibr" rid="B15">2023</xref>), speech recognition (Afouras et al., <xref ref-type="bibr" rid="B1">2018</xref>; Zhang et al., <xref ref-type="bibr" rid="B43">2019</xref>), and natural language processing (Shen et al., <xref ref-type="bibr" rid="B31">2018</xref>). However, it heavily relies on a large amount of labeled training data. When the available training data is drastically reduced, traditional deep learning methods are ineffective in training. In contrast, humans can quickly learn novel tasks (i.e., few-shot tasks) through a small amount of supervised information, because people can fully apply their past learning experience to novel tasks and then can quickly adapt and learn them. We hope that artificial intelligence models can quickly learn from novel tasks with few-shot data similar to humans. This fast learning is a challenge because the artificial intelligence models must combine their previous experience with a small amount of new information while avoiding over-fitting novel tasks (Finn et al., <xref ref-type="bibr" rid="B7">2017</xref>). The process of human learning has sparked our interest in the research of few-shot learning (Wang et al., <xref ref-type="bibr" rid="B38">2020</xref>; Lu et al., <xref ref-type="bibr" rid="B25">2023</xref>; Song et al., <xref ref-type="bibr" rid="B33">2023</xref>; Zeng and Xiao, <xref ref-type="bibr" rid="B41">2024</xref>) and how to fully utilize past learning experiences to few-shot tasks.</p>
<p>Meta-learning (Vanschoren, <xref ref-type="bibr" rid="B35">2018</xref>; Elsken et al., <xref ref-type="bibr" rid="B6">2020</xref>; Li et al., <xref ref-type="bibr" rid="B20">2021</xref>; Liu et al., <xref ref-type="bibr" rid="B24">2022</xref>; He et al., <xref ref-type="bibr" rid="B12">2023</xref>; Vettoruzzo et al., <xref ref-type="bibr" rid="B36">2024</xref>) was put forward to solve the problem of few-shot learning. It empowers learning systems with the ability to acquire knowledge from multiple tasks, enabling faster adaptation and generalization to new tasks (Vettoruzzo et al., <xref ref-type="bibr" rid="B36">2024</xref>). Specifically, it is to provide the model, especially the deep neural network, a learning ability that allows the model to learn some meta-knowledge automatically. Meta-knowledge refers to the knowledge that can be learned outside of the model training process, such as the initial parameters of the neural network, the structure and optimizer of the neural network, and the hyperparameter of the model. In few-shot learning, meta-learning specifically refers to learning meta-knowledge from a large number of prior tasks and using them to guide the model to learn faster in novel tasks.</p>
<p>The meta-learning method (Hospedales et al., <xref ref-type="bibr" rid="B13">2021</xref>), based on optimization, is an important branch of few-shot learning. These algorithms attempt to obtain a better initial model or correct gradient descent direction through meta-learning. It optimizes initial parameters by the meta-learner so that the learner can converge faster in novel tasks and achieve fast adaptation and learning with few-shot data.</p>
<p>In few-shot learning, the pre-learned base class data before learning few-shot novel tasks is crucial for the generalization ability of the model. Selecting a good base class can often greatly improve the learning efficiency of novel tasks (Zhou et al., <xref ref-type="bibr" rid="B45">2020</xref>). <xref ref-type="fig" rid="F1">Figure 1</xref> shows the randomly selected meta-tasks A and B from the base class data during the meta-training period. From the perspective of the sample category feature, meta-task B has more meta-knowledge related to the few-shot task. Therefore, it is crucial to select effective meta-knowledge from the base class for few-shot tasks, which can improve the efficiency and effect of few-shot learning.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The relationship between meta-tasks selected from base class dataset and few-shot dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-g0001.tif"/>
</fig>


<p>Based on the above motivation, this article argues that existing meta-learning methods do not take into account the relationship between base classes used for meta-learner learning and novel classes in few-shot tasks. During the meta-training period, this can lead to providing more irrelevant meta-knowledge for few-shot tasks, which will affect the efficiency and effect of few-shot learning in the meta-testing. Therefore, it is necessary to consider the feature relationships between base class data and few-shot data in the meta-training.</p>
<p>To improve this problem, first, we should select the most relevant set of candidate meta-tasks for novel tasks from the base class as much as possible to construct situational meta-tasks, which in turn provide optimal initial models for novel tasks. Then, the diversity of features among different meta-tasks can be used to promote better learning of novel tasks by multiple models.</p>
<p>In this article, we attempt to improve the problem of not considering relationships between base class data and few-shot novel class data by means of situational meta-task construction and multiple ensemble models. Starting from the feature relationship between base class data and few-shot data, we provide a new research idea for meta-learning. First of all, a universal feature extractor is trained in the basic learning phase to extract features of base class data and few-shot class data. Then, during the meta-training period, accurate meta-knowledge is provided for novel tasks by constructing situational meta-tasks. Furthermore, it provides good initial model parameters for novel tasks. Finally, during the meta-testing period, few-shot tasks are learned through the cooperation of multiple models. A large number of experiments on popular few-shot datasets demonstrate the effectiveness of the proposed method. Our main contributions are summarized as follows:</p>
<list list-type="bullet">
<list-item><p>This article puts forward a construction method of situational meta-task. It uses the class centroid of base class data to select the candidate meta-task sets with the stronger correlation for few-shot tasks and then sets up situational meta-tasks similar to few-shot tasks. The situational meta-tasks are the same as the few-shot tasks in form and similar to them in terms of feature. This method provides accurate and available meta-knowledge for novel tasks, which is conducive to rapid adaptation and learning on few-shot tasks.</p></list-item>
<list-item><p>An ensemble model method based on meta-optimization is proposed in this article. The meta-model of situational meta-task training during meta-training is used to cooperate to complete the learning of few-shot tasks. The cooperation of multiple models improves the predictive performance and stability of a single model in the full-phase meta-learning process.</p></list-item>
<list-item><p>Moreover, we extensively validate the proposed method by applying it to popular few-shot character dataset and image recognition datasets and then implementing and training through CNNs, Vgg16, and ResNet50 networks. The results indicate that the construction method of situational meta-task provides effective and available meta-knowledge for few-shot novel tasks, and the method of ensemble multiple models outperform previous state-of-the-art baselines.</p></list-item>
</list></sec>
<sec id="s2">
<title>2 Related work</title>
<p>The core idea of the meta-learning method is to use the past prior knowledge to guide the model to learn novel tasks. Meta-learning, as a standard approach to solving the problem of few-shot learning, which attempts to learn (Li et al., <xref ref-type="bibr" rid="B21">2017</xref>). The goal of meta-learning is to enable models, especially deep neural networks, to learn how to undertake novel tasks from few-shot data. Among them, the meta-learning method based on optimization is an important branch of few-shot learning.</p>
<sec>
<title>2.1 Meta-learning based on optimization</title>
<p>The idea of this kind of algorithm is to attempt to obtain a better initial model or correct gradient descent direction through meta-learning. Then, the initial parameters are optimized by the meta-learner. This enables the learner to converge faster in novel tasks and learn rapidly in the case of few-shot learning.</p>
<p>Finn et al. (<xref ref-type="bibr" rid="B7">2017</xref>) proposed a model agnostic meta-learning (MAML) method. First, the network is trained with the ability to extract universal features, and then further trained to adapt to novel tasks rapidly on this basis. This approach is considered model agnostic since it can be applied directly to any learning model trained by a gradient descent process.</p>
<p>Based on the idea of MAML, Li et al. (<xref ref-type="bibr" rid="B21">2017</xref>) put forward a meta-stochastic gradient descent method called meta-SGD based on LSTM. By meta-learning the initialization parameters, learning rate and updating direction, the trained model can be easily fine-tuned to adapt to novel tasks. This algorithm is significantly less difficult to train compared to LSTM. Compared with the MAML method, it improves the model capacity.</p>
<p>The reptile (Nichol et al., <xref ref-type="bibr" rid="B27">2018</xref>) method was proposed by Nichol et.al., which updates fewer parameters at a time and saves a lot of time and memory costs. However, the algorithm cannot directly adapt to the fast learning performed by MAML.</p>
<p>Rajeswaran et al. (<xref ref-type="bibr" rid="B29">2019</xref>) proposed a meta-learning method of implicit gradient. In this method, a new loss function and a corresponding method for computing the gradient are designed, so that the gradient of the parameter can be obtained only by computing the result of the loss function without considering its specific optimization method.</p>
<p>The meta-learning method based on optimization is to find a better initialization model or gradient descent direction for few-shot data. However, existing methods learn directly on the base class dataset and rarely consider relationships between few-shot data and base class data. This will lead to the learning of irrelevant meta-knowledge on the base class, which is not conducive to few-shot learning. Therefore, this article first fully considers the feature relationships between base class data and few-shot data and uses it as prior knowledge to construct strongly relevant situational meta-tasks for meta-learners. Then, meta-learners use situational meta-tasks to carry out meta-learning based on optimization, thereby improving the effect of few-shot learning.</p></sec>
<sec>
<title>2.2 Ensemble learning</title>
<p>To overcome the problem of unreliable and unstable results from single model, ensemble learning aims to utilize the diversity among multiple models to improve the learning ability of multiple weak learners. It can produce a strong ensemble learner for better prediction performance (Ganaie et al., <xref ref-type="bibr" rid="B10">2022</xref>).</p>
<p>Traditional ensemble learning methods include Bagging, Boosting, Stacking, decision tree-based, and random forest-based. The Bagging algorithm (Breiman, <xref ref-type="bibr" rid="B4">1996</xref>) (such as bootstrap aggregation) is one of the earliest ensemble learning methods. Although it has a simple structure, it has excellent performance. The algorithm generates different training subsets by randomly changing the distribution of the training dataset, then trains individual learners with different training subsets, and finally integrates them as a whole.</p>
<p>The Boosting algorithm (Freund and Schapire, <xref ref-type="bibr" rid="B9">1996</xref>) is an iteration method that transforms weak learners into a strong learner. It generates a strong learner that behaves almost perfectly by increasing the number of iterations. Stacking, also known as Stacked Generalization (Wolpert, <xref ref-type="bibr" rid="B39">1992</xref>), refers to training a model that is used to integrate all individual learners. The model is trained with the output of these individual learners as input to obtain a final output.</p>
<p>Recently, the deep neural network has been integrated into the ensemble strategies. Deep neural decision forest (Kontschieder et al., <xref ref-type="bibr" rid="B16">2015</xref>) is a learning method that combines convolutional neural networks (CNNs) and decision forest techniques. It introduces stochastic backpropagation of decision trees, which is then combined into a decision forest, resulting in a final model with better generalization performance. gcForest (Zhou and Feng, <xref ref-type="bibr" rid="B46">2017</xref>) is a new method that combines the ensemble method with the deep neural network. Unlike the above method, it replaces the neurons with random forest models, using the output vector of each random forest as the input to the next layer.</p></sec></sec>
<sec id="s3">
<title>3 Method</title>
<p>In this section, first, we explain the relevant definitions and concepts proposed for meta-task construction and meta-model ensemble (Section 3.1). Then, the motivation and idea of our proposed method are generally introduced (Section 3.2). Finally, in order to improve the problem of feature correlation between few-shot and base class, the <bold>c</bold>onstruction method of <bold>s</bold>ituational <bold>m</bold>eta-<bold>t</bold>ask (CSMT) proposed in meta-training (Section 3.3) is introduced and the <bold>f</bold>ull-phase <bold>m</bold>eta-learning <bold>p</bold>rocess of <bold>m</bold>ultiple <bold>i</bold>nitial <bold>m</bold>odel <bold>c</bold>ooperation (FMPMIMC) is described, which includes the basic learning phase and the meta-optimization phase (Section 3.4).</p>
<sec>
<title>3.1 Problem definition and description</title>
<sec>
<title>3.1.1 Basic class dataset and novel class dataset(few-shot dataset)</title>
<p>The base class dataset is <inline-formula><mml:math id="M1"><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, and the novel class dataset is <inline-formula><mml:math id="M2"><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, where <italic>D</italic><sub><italic>base</italic></sub>&#x02229;<italic>D</italic><sub><italic>novel</italic></sub> &#x0003D; &#x02205;. Few-shot tasks <italic>T</italic><sub><italic>novel</italic></sub> are randomly sampled from <italic>D</italic><sub><italic>novel</italic></sub>. Each few-shot task includes support set and query set. The support set is <inline-formula><mml:math id="M3"><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, where <italic>S</italic>&#x02282;<italic>T</italic><sub><italic>novel</italic></sub>. The query set is <inline-formula><mml:math id="M4"><mml:mi>Q</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>q</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, where <italic>Q</italic>&#x02282;<italic>T</italic><sub><italic>novel</italic></sub>. Especially, <italic>S</italic>&#x02229;<italic>Q</italic> &#x0003D; &#x02205; and <italic>S</italic>&#x022C3;<italic>Q</italic> &#x0003D; <italic>T</italic><sub><italic>novel</italic></sub>.</p></sec>
<sec>
<title>3.1.2 Situational meta-task</title>
<p>Situational meta-task is a collection of tasks that have the same form (N-way K-shot) and related features to few-shot tasks. It is constructed from the data in the base class and is used in the meta-training. From the perspective of feature, it has a strong correlation with few-shot tasks. In form, it is the same as Nway-Kshot for few-shot tasks. Suppose that the base class dataset is represented as <italic>D</italic><sub><italic>base</italic></sub> &#x0003D; {<italic>D</italic><sub>1</sub>, <italic>D</italic><sub>2</sub>, <italic>D</italic><sub>3</sub>, ..., <italic>D</italic><sub><italic>n</italic></sub>} by category, and a 5way-1shot support set denotes <italic>S</italic> &#x0003D; {<italic>x</italic><sub>1</sub>, <italic>x</italic><sub>2</sub>, <italic>x</italic><sub>3</sub>, <italic>x</italic><sub>4</sub>, and <italic>x</italic><sub>5</sub>}. By using the situational meta-task construction method, the most relevant candidate meta-task set (such as the candidate meta-task set for few-shot <italic>x</italic><sub>1</sub> is <inline-formula><mml:math id="M5"><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x000A0;and&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, where <italic>D</italic><sub><italic>c</italic></sub>, <italic>D</italic><sub><italic>k</italic></sub>, <italic>D</italic><sub><italic>p</italic></sub>, <italic>D</italic><sub><italic>q</italic></sub>, and <italic>D</italic><sub><italic>m</italic></sub>&#x02208;<italic>D</italic><sub><italic>base</italic></sub>) is selected from the base class dataset for each category of few-shot dataset (taking few-shot <italic>x</italic><sub>1</sub> as an example), and then the situational meta-tasks are constructed by extracting a sample from each category of few-shot own related candidate meta-task set (such as <inline-formula><mml:math id="M6"><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:math></inline-formula> <inline-formula><mml:math id="M7"><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x000A0;and&#x000A0;</mml:mtext><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, where <inline-formula><mml:math id="M8"><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:munder class="msub"><mml:mrow><mml:mo>&#x022C2;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02260;</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:munder><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mo class="MathClass-ord">&#x02205;</mml:mo></mml:math></inline-formula>).</p></sec>
<sec>
<title>3.1.3 Multiple initial models</title>
<p>During the meta-training period, the set of different meta-models is trained using different situational meta-tasks, which can be described as <italic>M</italic> &#x0003D; {<italic>M</italic><sub>1</sub>, <italic>M</italic><sub>2</sub>, <italic>M</italic><sub>3</sub>, ..., <italic>M</italic><sub><italic>n</italic></sub>}. They are used as the basis for cooperative learning in the meta-testing.</p></sec>
<sec>
<title>3.1.4 Full-phase meta-learning process</title>
<p>It includes the basic learning phase and the meta-optimization phase. The basic learning phase provides a universal feature extractor for the meta-optimization phase, which is used for constructing situational meta-tasks. The meta-optimization phase includes meta-training and meta-testing. During the meta-training period, the multiple initial models are trained using situational meta-tasks. They are used to adapt and learn few-shot tasks cooperatively in the meta-testing period.</p></sec></sec>
<sec>
<title>3.2 Overview</title>
<p>The full-phase meta-learning method based on situational meta-task construction and multiple initial model cooperation is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. It consists of the basic learning phase and the meta-optimization phase. The base class data and few-shot data do not have the same category, which means that few-shot data are novel tasks for models. The general idea of our proposed method is introduced below from the process of the full-phase meta-learning.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Schematic diagram of a full-phase meta-learning method based on construction of situational meta-task and cooperation with multiple initial models.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-g0002.tif"/>
</fig>


<p>First, in the basic learning phase, a universal feature extractor is trained for constructing situational meta-tasks in the meta-optimization phase. The purpose of constructing situational meta-tasks is to provide more effective meta-knowledge for few-shot tasks, enabling the model to rapidly adapt to few-shot tasks. Then, situational meta-tasks are used to train multiple initial models in the meta-training of the meta-optimization phase. Finally, in the meta-testing of the meta-optimization phase, the few-shot dataset is used to optimize multiple initial models, promoting cooperation among models, and more fully utilizing meta-knowledge to learn few-shot classification model.</p></sec>
<sec>
<title>3.3 A construction method of situational meta-task</title>
<p>The meta-learning methods based on optimization find better initial models or gradient descent directions for few-shot tasks through base class dataset. This allows models to adapt and learn quickly for few-shot tasks. However, existing methods directly learn on the base class dataset, rarely considering the feature relationships between few-shot data and base class data. This will result in learning more irrelevant meta-knowledge on the base class, which is not conducive to few-shot learning. Therefore, our research motivation is to provide relevant and effective meta-knowledge for few-shot tasks from the base class. Furthermore, it provides better initial model parameters for few-shot learning, which enables fast learning and adaptation on few-shot data.</p>
<p>In order to solve the above problem, this article proposes a construction method of situational meta-task (CSMT), which is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. The main idea of this method is to select the categories related to few-shot tasks from the base class dataset as candidate meta-task sets, and then use candidate meta-task sets to construct situational meta-tasks. Specifically, first, few-shot tasks <italic>T</italic><sub><italic>novel</italic></sub> (<xref ref-type="fig" rid="F3">Figure 3E</xref>) are randomly sampled from the few-shot dataset <italic>D</italic><sub><italic>novel</italic></sub> (<xref ref-type="fig" rid="F3">Figure 3D</xref>). Each few-shot task consists of a support set (<italic>S</italic><sub>1</sub>, <italic>S</italic><sub>2</sub>, ...) and a query set (<italic>Q</italic><sub>1</sub>, <italic>Q</italic><sub>2</sub>, ...). Then, the support set <italic>S</italic><sub>1</sub> (shown in the above <xref ref-type="fig" rid="F3">Figure 3E</xref>) of 5-way 1-shot task 1 is used as an example to construct its situational meta-tasks. The relevant categories from the base class dataset <italic>D</italic><sub><italic>base</italic></sub> (<xref ref-type="fig" rid="F3">Figure 3A</xref>) is selected as a candidate meta-task set using the feature relationships between the centroid <inline-formula><mml:math id="M9"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> of each category of the base class dataset and the centroid <inline-formula><mml:math id="M10"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> of each category of the support set <italic>S</italic><sub>1</sub>. The relevant categories from the base class dataset are selected as the candidate meta-task set <italic>Meta</italic>_<italic>task</italic><sub><italic>S</italic><sub>1</sub></sub> (<xref ref-type="fig" rid="F3">Figure 3B</xref>). Finally, the candidate meta-task set is used to construct some situational meta-tasks <inline-formula><mml:math id="M11"><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> (<xref ref-type="fig" rid="F3">Figure 3C</xref>) for the 5-way 1-shot task 1. The situational meta-task A has the same form(5-way 1-shot) and related features to the support set <italic>S</italic><sub>1</sub> of the 5-way 1-shot task 1. The situational meta-tasks are used during meta-training and the few-shot tasks are used during meta-testing.</p>












<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>A schematic diagram of the situational meta-task construction process. <bold>(A)</bold> Base class dataset (<italic>D</italic><sub><italic>base</italic></sub>). <bold>(B)</bold> Candidate meta-task set (<italic>Meta</italic>_<italic>task</italic><sub><italic>S</italic><sub>1</sub></sub>). <bold>(C)</bold> Situational meta-task (<inline-formula><mml:math id="M12"><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>). <bold>(D)</bold> Few-shot database (<italic>D</italic><sub><italic>novel</italic></sub>). <bold>(E)</bold> Few-shot task (<italic>T</italic><sub><italic>novel</italic></sub>).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-g0003.tif"/>
</fig>


<p><bold>The construction process of situational meta-task is given as follows:</bold></p>
<p><bold>Step 1: Computing the central support point (centroid) of each class</bold></p>
<p>The mean vector is computed for all feature vectors of each class in the base class dataset as the central support point <inline-formula><mml:math id="M13"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> for that class. It can be represented as <xref ref-type="disp-formula" rid="E1">Equation (1)</xref>:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>base</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>base</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>base</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>s</mml:mtext></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mtext>y</mml:mtext></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>base</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>s</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>base</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C6;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>base</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>s</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M15"><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the sample set of the ith class in the base class dataset and <inline-formula><mml:math id="M16"><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the feature vector that belongs to <inline-formula><mml:math id="M17"><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>. <italic>f</italic><sub>&#x003C6;</sub>(&#x000B7;) is an embedding function.</p>
<p>Similarly, when the form of the few-shot dataset is N-way K-shot, the mean vector of all feature vectors in each category in the few-shot data set is calculated as the central support point <inline-formula><mml:math id="M18"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> of the class. When the form of the few-shot dataset is N-way 1-shot, the central support point <inline-formula><mml:math id="M19"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> of each category is the sample feature. It can be represented as <xref ref-type="disp-formula" rid="E2">Equation (2)</xref>:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>novel</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>K</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>novel</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>s</mml:mtext></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mtext>y</mml:mtext></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>novel</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>s</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>novel</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C6;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>novel</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>s</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><bold>Step 2: Selecting few-shot candidate meta-task sets from the base class dataset</bold></p>
<p>The feature distance (<italic>Dis</italic><sub><italic>j</italic>_<italic>i</italic></sub>) between the central support point <inline-formula><mml:math id="M21"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> of each category in the few-shot dataset and the central support point <inline-formula><mml:math id="M22"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> of each category in the base class dataset are calculated, and the distance using cosine similarity is calculated. It can be represented as <xref ref-type="disp-formula" rid="E3">Equation (3)</xref>:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mo class="qopname">Dis</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo></mml:mrow></mml:msub><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo class="qopname">cos</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>novel</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>base</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The calculated data are sorted in the descending order, and the top K class is selected as the candidate meta-task set for each few-shot class and is denoted as <italic>Meta</italic>_<italic>task</italic><sub><italic>j</italic></sub>. It can be represented as <xref ref-type="disp-formula" rid="E4">Equation (4)</xref>:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>M</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mtext>_</mml:mtext><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02190;</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mtext>_</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mo>,</mml:mo><mml:mi>K</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><bold>Step 3: Handling the conflict of candidate meta-task sets</bold></p>
<p>When two or more candidate meta-task sets contain the same category in the base class (assuming that the candidate meta-task sets corresponding to classes p and q of the few-shot dataset both contain class m of the base class dataset), select the few-shot class with the minimum centroid error as the optimal construction method to ensure that different few-shot classes select different candidate meta-task sets. The centroid error is the sum of the distance between all samples of a certain class in the few-shot dataset and the centroid of that class in the base class dataset. It can be represented as <xref ref-type="disp-formula" rid="E5">Equation (5)</xref>:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>novel</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mrow><mml:mtext>D</mml:mtext></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>novel</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msubsup></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C6;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mtext>x</mml:mtext></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>novel</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Among them, <italic>L</italic><sub><italic>c</italic></sub> represents the centroid error between the samples <inline-formula><mml:math id="M26"><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> of the class p in the few-shot dataset and the class m in the base class dataset. <inline-formula><mml:math id="M27"><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> represents the sample in the few-shot dataset <inline-formula><mml:math id="M28"><mml:msubsup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M29"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the centroid of the class m in the base class dataset.</p>
<p><bold>Step 4: Constructing situational meta-tasks</bold></p>
<p>The tasks from candidate meta-task sets for each class of few-shot data are extracted and combined into situational meta-tasks in the form of Nway-Kshot which is the same as few-shot tasks. They are the training dataset in the meta-training period.</p>
<p>The construction method of situational meta-task is shown in <xref ref-type="table" rid="T5">Algorithm 1</xref>.</p>
<table-wrap position="float" id="T5">
<label>Algorithm 1</label>
<caption><p>The CSMT method.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-i0001.tif"/>
</table-wrap>
<p>Through the situational meta-task construction method, the training dataset related to the feature of few-shot is provided for the meta-model in the meta-training. First, for each few-shot class, strongly related candidate meta-task sets are selected from the base class dataset in order to better provide useful meta-knowledge for few-shot data. Then, the candidate meta-task sets are used to construct situational meta-tasks, whose form and features are more similar to few-shot tasks, which is beneficial for the model to adapt quickly and learn novel tasks.</p>
<p>In this subsection, different situational meta-tasks provide models containing different meta-knowledge for few-shot tasks. Overall, this process also makes full preparation for the efficient learning of few-shot tasks in the next subsection.</p></sec>
<sec>
<title>3.4 Full-phase meta-learning process based on multiple initial model cooperation</title>
<p>As shown in <xref ref-type="fig" rid="F4">Figure 4</xref>, the full-phase meta-learning process based on multiple initial model cooperation (FMPMIMC) includes two phases: basic learning and meta-optimization. The basic learning phase provides a universal feature extractor for constructing situational meta-tasks in the meta-optimization phase.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>A schematic diagram of the learning process of a full-phase meta-learning method based on situational meta-task construction and cooperation with multiple initial models. In the basic training phase (above), the model learns a universal feature extractor from the base class data for situational meta-task construction. In the meta-optimization phase (below), multiple independent models are trained by situational meta-tasks in the meta-training. Then, multiple models utilize classification loss and cooperative loss to learn few-shot novel tasks in the meta-testing.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-g0004.tif"/>
</fig>

<sec>
<title>3.4.1 The basic learning phase</title>
<p>The model is trained by the base class data, and it can be described as <xref ref-type="disp-formula" rid="E6">Equation (6)</xref>:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M38"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mo>&#x000B0;</mml:mo><mml:mi>w</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi mathvariant='double-struck'>E</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>w</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where <italic>L</italic><sub><italic>base</italic>_<italic>cls</italic></sub> is the classification loss, <italic>f</italic>(&#x000B7;) is a feature extractor of the model, <italic>w</italic>(&#x000B7;) is a classifier, and <italic>l</italic>(&#x000B7;, &#x000B7;) is a cross entropy loss function.</p>
<p>The basic learning phase can be analogized to the extensive human learning process, and the model gets a universal feature extractor through extensive learning. It is better to extract features in the meta-optimization phase.</p></sec>
<sec>
<title>3.4.2 The meta-optimization phase</title>
<p>The meta-optimization phase includes two interactive processes: meta-training and meta-testing. First, during the meta-training period, some situational meta-tasks are constructed for few-shot tasks using the feature extractor from the base learning process (each situational meta-task contains the corresponding support set and query set).</p>
<p>Then, they are used to train several independent networks (each network includes components such as feature extractor, meta-learner, and classifier). The loss of each network utilizes the classified cross entropy loss of situational meta-tasks, which can be represented as <xref ref-type="disp-formula" rid="E7">Equation (7)</xref>:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M39"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mo>&#x000B0;</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x000B0;</mml:mo><mml:mi>w</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mi mathvariant='double-struck'>E</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>w</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>L</italic><sub><italic>meta</italic>_<italic>train</italic></sub> is the meta-training loss of situational meta-tasks and <italic>f</italic>(&#x000B7;) is the feature extractor of the network. <italic>m</italic>(&#x000B7;) is the meta-learner and <italic>w</italic>(&#x000B7;) is the classifier. <italic>l</italic>(&#x000B7;, &#x000B7;) is the cross entropy loss function of situational meta-task.</p>
<p>During the meta-training period, models containing diverse meta-knowledge are trained and learned on some different situational meta-tasks. During the meta-testing period, multiple initial model cooperation is used to learn novel few-shot tasks. The single model utilizes traditional cross entropy function to calculate the classification loss of the support sets in novel tasks. It can be represented as <xref ref-type="disp-formula" rid="E8">Equation (8)</xref>:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M40"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:msubsup><mml:mi>L</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x000B0;</mml:mo><mml:msub><mml:mi>m</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x000B0;</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mi>D</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mi mathvariant='double-struck'>E</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>m</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mo>&#x0005F;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M42"><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the classification loss of a single model learning novel few-shot tasks, <inline-formula><mml:math id="M43"><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the support set sample of few-shot task, and <inline-formula><mml:math id="M44"><mml:msubsup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the label of the support set sample of few-shot task.</p>
<p>In order to cooperate with multiple initial models to complete the learning of few-shot novel tasks, KL divergence is used between the models to calculate the difference loss in model predictions. By strengthening cooperation among models, efficient learning of few-shot novel tasks can be realized. The cooperation loss of multiple initial models can be described as <xref ref-type="disp-formula" rid="E9">Equation (9)</xref>:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M45"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mtext>_</mml:mtext><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mtext>_</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>K</mml:mi><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M46"><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the difference loss predicted between the ith model and other models, and <inline-formula><mml:math id="M47"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the output of the softmax layer of the ith network.</p>
<p>During the meta-testing period, the total loss of the ith model is as follows</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M48"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mtext>_</mml:mtext><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mtext>_</mml:mtext><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mtext>_</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mtext>_</mml:mtext><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mtext>_</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M49"><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the total loss when the ith model learns few-shot tasks. <inline-formula><mml:math id="M50"><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the classification loss, and <inline-formula><mml:math id="M51"><mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the cooperation loss between models.</p>
<p>The full-phase meta-learning process based on multiple initial model cooperation is shown in <xref ref-type="table" rid="T6">Algorithm 2</xref>.</p>
<table-wrap position="float" id="T6">
<label>Algorithm 2</label>
<caption><p>The FMPMIMC method.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-i0002.tif"/>
</table-wrap>
<p>In this section, a situational meta-task construction method is used to provide more relevant and effective meta-knowledge for few-shot novel tasks. During the meta-training period, different situational meta-tasks provide diverse meta-knowledge, resulting in certain differences in the meta-knowledge learned by models. During the meta-testing period, the diversity of models is used for cooperative learning few-shot novel tasks. By reducing the prediction differences among models, the prediction quality and stability of the whole model are further improved.</p></sec></sec></sec>
<sec id="s4">
<title>4 Experiments</title>
<p>In this section, first, we introduce several benchmark few-shot datasets (Omniglot, CIFAR-100 and MiniImageNet) used in our experiments (Section 4.1). Then, we conduct three experiments, namely, the situational meta-task construction experiment (Section 4.2), the classification experiment of the full-phase meta-learning multiple initial model cooperation (Section 4.3), and the related parameter setting experiment (Section 4.4). These experiments are used to evaluate CSMT and FMPMIMC methods.</p>
<sec>
<title>4.1 Setup</title>
<p>In this study, the performance of the proposed method is evaluated on three few-shot image classification datasets, including the simple character dataset Omniglot (Lake et al., <xref ref-type="bibr" rid="B17">2011</xref>), the complex image datasets CIFAR-100 (Boris et al., <xref ref-type="bibr" rid="B3">2018</xref>), and MiniImageNet (Vinyals et al., <xref ref-type="bibr" rid="B37">2016</xref>). <xref ref-type="fig" rid="F5">Figure 5</xref> shows typical images for each dataset.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>The samples of the standard dataset used in the experiments.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-g0005.tif"/>
</fig>


<sec>
<title>4.1.1 Omniglot dataset</title>
<p>It consists of 1,623 handwritten characters (equivalent to 1,623 classes) from 50 different languages. Each class has 20 different handwritings (equivalent to 20 samples in each class). The size of each sample is 28 &#x000D7; 28 pixels. In each few-shot experiment, we randomly selected 100 classes as the few-shot dataset (five classes were selected multiple times as few-shot tasks), and the remaining classes were used as the base class dataset.</p></sec>
<sec>
<title>4.1.2 CIFAR-100 dataset</title>
<p>It contains 100 classes, each with 600 color images of size 32 &#x000D7; 32 pixels. In total, 500 samples from each class are used as the training dataset, and the remaining 100 samples are used as the testing dataset. In each few-shot experiment, we randomly selected 20 classes as the few-shot dataset(5 classes were selected multiple times as few-shot tasks), and the remaining 80 classes were used as the base class dataset.</p></sec>
<sec>
<title>4.1.3 MiniImageNet dataset</title>
<p>It consists of 100 classes selected from ImageNet, and each category has 600 color images with the size of 84 &#x000D7; 84 pixels. Among them, the training dataset, the validation dataset, and the testing dataset contain 64 classes, 16 classes and 20 classes, respectively. In each few-shot experiment, we randomly selected five classes from the testing dataset multiple times as few-shot learning tasks, and the remaining 80 classes from the training and validation datasets as the base class dataset.</p></sec></sec>
<sec>
<title>4.2 The experiment of situational meta-task construction</title>
<p>For the construction of situational meta-task in the experiment, first, we randomly selected 1 or 5 samples from five classes as few-shot tasks (5way-1shot/5way-5shot) from the few-shot dataset for each experiment. Then, during the meta-training period, the construction method of situational meta-task is used to select candidate meta-task sets for few-shot novel tasks. Using the experiment results from the Omniglot dataset as an illustration, <xref ref-type="fig" rid="F6">Figure 6</xref> shows an example of candidate meta-task sets selected according to the 5way-1shot task. The following are the analysis of the experimental results.</p>





<list list-type="simple">
<list-item><p>(1) The candidate meta-task set selected for the 5way-1shot task in the Omniglot dataset is visualized in <xref ref-type="fig" rid="F6">Figure 6</xref>. The features and shapes of sample classes in the candidate meta-task set are similar to those of few-shot task, and its samples can provide more useful and effective meta-knowledge for the few-shot task. Then samples are extracted corresponding to the few-shot task form (5way-1shot) to construct situational meta-tasks.</p></list-item>
<list-item><p>(2) The average accuracy of the 5way-1shot and 5way-5shot experiments using a single model on the Omniglot dataset by CSMT is reported in <xref ref-type="table" rid="T1">Table 1</xref>. In the 5way-1shot experiment of the Omniglot dataset, the experimental result is 0.23% higher than that of the advanced SNAIL method. This shows that the CSMT method provides more effective meta-knowledge for few-shot tasks, which is helpful for few-shot learning.</p>
</list-item>
<list-item><p>(3) <xref ref-type="table" rid="T2">Table 2</xref> shows the average accuracy of 5way-1shot and 5way-5shot experiments of a single model on the CIFAR-100 dataset by the CSMT method. Compared with the advanced Dual TriNet method, it improves the performance by 6.16% and 1.1%. The experimental results demonstrate the effectiveness of the meta-knowledge provided for few-shot tasks during the meta-training period. The performance is outstanding in the experiment of 5-way 1-shot, which shows that the model can rapidly adapt to the learning of few-shot tasks through the training of situational meta-tasks.</p></list-item>
<list-item><p>(4) <xref ref-type="fig" rid="F7">Figure 7</xref> shows the ablation experiment of the CSMT method. The three cases of providing situational meta-tasks, selecting random meta-tasks, and providing irrelevant meta-tasks for the meta-model are compared. It shows the learning effect of the model on the novel task as the number of iterations increases in the meta-testing. As can be seen from <xref ref-type="fig" rid="F7">Figure 7</xref>, it is important to provide effective meta-knowledge for few-shot tasks. The CSMT method enables the model to adapt to few-shot tasks more quickly.</p></list-item>
</list>

<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>The samples of candidate meta-task set selected for the few-shot task in the Omniglot dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-g0006.tif"/>
</fig>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>The 5-way 1-shot /5-shot CSMT classification accuracy (%) on the Omniglot dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>5-way 1-shot</bold></th>
<th valign="top" align="left"><bold>5-way 5-shot</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">MAML (Finn et al., <xref ref-type="bibr" rid="B7">2017</xref>)</td>
<td valign="top" align="left">98.7 &#x000B1; 0.4</td>
<td valign="top" align="left"><bold>99.9</bold> &#x000B1; <bold>0.1</bold></td>
</tr> <tr>
<td valign="top" align="left">TCML (Mishra et al., <xref ref-type="bibr" rid="B26">2017</xref>)</td>
<td valign="top" align="left">98.96 &#x000B1; 0.2</td>
<td valign="top" align="left">99.75 &#x000B1; 0.11</td>
</tr> <tr>
<td valign="top" align="left">Gaussian PN (Fort, <xref ref-type="bibr" rid="B8">2017</xref>)</td>
<td valign="top" align="left">99.07 &#x000B1; 0.03</td>
<td valign="top" align="left">99.73 &#x000B1; 0.02</td>
</tr> <tr>
<td valign="top" align="left">Reptile (Nichol et al., <xref ref-type="bibr" rid="B27">2018</xref>)</td>
<td valign="top" align="left">97.68 &#x000B1; 0.04</td>
<td valign="top" align="left">99.48 &#x000B1; 0.06</td>
</tr> <tr>
<td valign="top" align="left">SNAIL (Nikhil et al., <xref ref-type="bibr" rid="B28">2018</xref>)</td>
<td valign="top" align="left">99.07 &#x000B1; 0.16</td>
<td valign="top" align="left">99.78 &#x000B1; 0.09</td>
</tr> <tr>
<td valign="top" align="left">R2-D2 (Bertinetto et al., <xref ref-type="bibr" rid="B2">2019</xref>)</td>
<td valign="top" align="left">98.91 &#x000B1; 0.05</td>
<td valign="top" align="left">99.74 &#x000B1; 0.02</td>
</tr> <tr>
<td valign="top" align="left">CSMT (ours)</td>
<td valign="top" align="left"><bold>99.3</bold> &#x000B1; <bold>0.18</bold></td>
<td valign="top" align="left">99.6 &#x000B1; 0.12</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values are the best experimental results.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>The 5-way 1-shot /5-shot CSMT classification accuracy (%) on the CIFAR-100 dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>5-way 1-shot</bold></th>
<th valign="top" align="left"><bold>5-way 5-shot</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">TADAM (Boris et al., <xref ref-type="bibr" rid="B3">2018</xref>)</td>
<td valign="top" align="left">40.10 &#x000B1; 0.40</td>
<td valign="top" align="left">56.10 &#x000B1; 0.40</td>
</tr> <tr>
<td valign="top" align="left">MetaOptNet (Lee et al., <xref ref-type="bibr" rid="B19">2019</xref>)</td>
<td valign="top" align="left">41.10 &#x000B1; 0.60</td>
<td valign="top" align="left">55.50 &#x000B1; 0.60</td>
</tr> <tr>
<td valign="top" align="left">ProtoNet (Snell et al., <xref ref-type="bibr" rid="B32">2017</xref>)</td>
<td valign="top" align="left">41.54 &#x000B1; 0.76</td>
<td valign="top" align="left">57.08 &#x000B1; 0.76</td>
</tr> <tr>
<td valign="top" align="left">DC (Lifchitz et al., <xref ref-type="bibr" rid="B22">2019</xref>)</td>
<td valign="top" align="left">42.04 &#x000B1; 0.17</td>
<td valign="top" align="left">57.05 &#x000B1; 0.16</td>
</tr> <tr>
<td valign="top" align="left">Matching Nets (Vinyals et al., <xref ref-type="bibr" rid="B37">2016</xref>)</td>
<td valign="top" align="left">43.88 &#x000B1; 0.75</td>
<td valign="top" align="left">57.05 &#x000B1; 0.71</td>
</tr> <tr>
<td valign="top" align="left">MTL (Sun et al., <xref ref-type="bibr" rid="B34">2019</xref>)</td>
<td valign="top" align="left">45.10 &#x000B1; 1.80</td>
<td valign="top" align="left">57.60 &#x000B1; 0.90</td>
</tr> <tr>
<td valign="top" align="left">DeepEMD (Zhang et al., <xref ref-type="bibr" rid="B42">2022</xref>)</td>
<td valign="top" align="left">46.47 &#x000B1; 0.78</td>
<td valign="top" align="left">63.22 &#x000B1; 0.71</td>
</tr> <tr>
<td valign="top" align="left">DEML&#x0002B;Meta-SGD (Zhou et al., <xref ref-type="bibr" rid="B44">2018</xref>)</td>
<td valign="top" align="left">61.62 &#x000B1; 1.01</td>
<td valign="top" align="left">77.94 &#x000B1; 0.74</td>
</tr> <tr>
<td valign="top" align="left">Dual TriNet (Chen et al., <xref ref-type="bibr" rid="B5">2019</xref>)</td>
<td valign="top" align="left">63.41 &#x000B1; 0.64</td>
<td valign="top" align="left">78.43 &#x000B1; 0.62</td>
</tr> <tr>
<td valign="top" align="left">CSMT (ours)</td>
<td valign="top" align="left"><bold>69.57</bold> &#x000B1; <bold>1.20</bold></td>
<td valign="top" align="left"><bold>79.53</bold> &#x000B1; <bold>0.93</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values are the best experimental results.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>A comparison of the training process of ablation experiments for the CSMT method based on CIFAR-100.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-g0007.tif"/>
</fig>
<p>Through the construction method of situational meta-task, useful meta-knowledge is provided for the learning of novel tasks in the meta-testing. However, different situational meta-tasks provide different meta-knowledge for few-shot tasks. In order to make full use of the meta-knowledge of different situational meta-tasks, a full-phase meta-learning experiment with multiple initial model cooperation is carried out in the next section.</p></sec>
<sec>
<title>4.3 The experiment of the full-phase meta-learning process based on multiple initial model cooperation</title>
<p>In the previous subsection, the CSMT method was used to provide multiple initial models for learning few-shot novel tasks in the meta-testing. There are differences in the meta-knowledge provided by different initial models, resulting in different learning effects on novel tasks. In this subsection, the FMPMIMC method is used to reduce the differences between models and realize the rapid adaptation and efficient learning of few-shot novel tasks.</p>
<p>First, the CSMT method is used to provide <italic>n</italic> initial meta-models (<italic>n</italic> &#x0003D; 2, 3, 5, 10, and 20) for few-shot novel tasks. Then, during the meta-training period, each of the <italic>n</italic> initial meta-models uses its own situational meta-tasks training. During the meta-testing period, these <italic>n</italic> initial meta-models are trained together on corresponding few-shot tasks by the FMPMIMC method. Finally, we conduct 5way-1shot and 5way-5shot experiments on CIFAR-100 and MiniImageNet datasets, respectively. The average results of the experiments are reported as shown in <xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref>. The following are the analysis of the experimental results.</p>






<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>The 5-way 1-shot /5-shot FMPMIMC classification accuracy (%) on the CIFAR-100 dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>5-way 1-shot</bold></th>
<th valign="top" align="left"><bold>5-way 5-shot</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">TADAM (Boris et al., <xref ref-type="bibr" rid="B3">2018</xref>)</td>
<td valign="top" align="left">40.10 &#x000B1; 0.40</td>
<td valign="top" align="left">56.10 &#x000B1; 0.40</td>
</tr> <tr>
<td valign="top" align="left">MetaOptNet (Lee et al., <xref ref-type="bibr" rid="B19">2019</xref>)</td>
<td valign="top" align="left">41.10 &#x000B1; 0.60</td>
<td valign="top" align="left">55.50 &#x000B1; 0.60</td>
</tr> <tr>
<td valign="top" align="left">ProtoNet (Snell et al., <xref ref-type="bibr" rid="B32">2017</xref>)</td>
<td valign="top" align="left">41.54 &#x000B1; 0.76</td>
<td valign="top" align="left">57.08 &#x000B1; 0.76</td>
</tr> <tr>
<td valign="top" align="left">DC (Lifchitz et al., <xref ref-type="bibr" rid="B22">2019</xref>)</td>
<td valign="top" align="left">42.04 &#x000B1; 0.17</td>
<td valign="top" align="left">57.05 &#x000B1; 0.16</td>
</tr> <tr>
<td valign="top" align="left">Matching Nets (Vinyals et al., <xref ref-type="bibr" rid="B37">2016</xref>)</td>
<td valign="top" align="left">43.88 &#x000B1; 0.75</td>
<td valign="top" align="left">57.05 &#x000B1; 0.71</td>
</tr> <tr>
<td valign="top" align="left">MTL (Sun et al., <xref ref-type="bibr" rid="B34">2019</xref>)</td>
<td valign="top" align="left">45.10 &#x000B1; 1.80</td>
<td valign="top" align="left">57.60 &#x000B1; 0.90</td>
</tr> <tr>
<td valign="top" align="left">DeepEMD (Zhang et al., <xref ref-type="bibr" rid="B42">2022</xref>)</td>
<td valign="top" align="left">46.47 &#x000B1; 0.78</td>
<td valign="top" align="left">63.22 &#x000B1; 0.71</td>
</tr> <tr>
<td valign="top" align="left">DEML&#x0002B;Meta-SGD (Zhou et al., <xref ref-type="bibr" rid="B44">2018</xref>)</td>
<td valign="top" align="left">61.62 &#x000B1; 1.01</td>
<td valign="top" align="left">77.94 &#x000B1; 0.74</td>
</tr> <tr>
<td valign="top" align="left">Dual TriNet (Chen et al., <xref ref-type="bibr" rid="B5">2019</xref>)</td>
<td valign="top" align="left">63.41 &#x000B1; 0.64</td>
<td valign="top" align="left">78.43 &#x000B1; 0.62</td>
</tr> <tr>
<td valign="top" align="left">CSMT</td>
<td valign="top" align="left">69.57 &#x000B1; 1.20</td>
<td valign="top" align="left">79.53 &#x000B1; 0.93</td>
</tr> <tr>
<td valign="top" align="left">FMPMIMC (without KL divergence)</td>
<td valign="top" align="left">70.40 &#x000B1; 0.65</td>
<td valign="top" align="left">80.82 &#x000B1; 0.70</td>
</tr> <tr>
<td valign="top" align="left">FMPMIMC (ours)</td>
<td valign="top" align="left"><bold>73.15</bold> &#x000B1; <bold>0.53</bold></td>
<td valign="top" align="left"><bold>83.06</bold> &#x000B1; <bold>0.60</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values are the best experimental results.</p>
</table-wrap-foot>
</table-wrap>

<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>The 5-way 1-shot /5-shot FMPMIMC classification accuracy (%) on MiniImagenet dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>5-way 1-shot</bold></th>
<th valign="top" align="left"><bold>5-way 5-shot</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Matching Nets (Vinyals et al., <xref ref-type="bibr" rid="B37">2016</xref>)</td>
<td valign="top" align="left">43.56 &#x000B1; 0.84</td>
<td valign="top" align="left">55.31 &#x000B1; 0.73</td>
</tr> <tr>
<td valign="top" align="left">ProtoNet (Snell et al., <xref ref-type="bibr" rid="B32">2017</xref>)</td>
<td valign="top" align="left">49.42 &#x000B1; 0.78</td>
<td valign="top" align="left">68.20 &#x000B1; 0.66</td>
</tr> <tr>
<td valign="top" align="left">Dual TriNet (Chen et al., <xref ref-type="bibr" rid="B5">2019</xref>)</td>
<td valign="top" align="left">58.12 &#x000B1; 1.37</td>
<td valign="top" align="left">76.92 &#x000B1; 0.69</td>
</tr> <tr>
<td valign="top" align="left">DEML&#x0002B;Meta-SGD (Zhou et al., <xref ref-type="bibr" rid="B44">2018</xref>)</td>
<td valign="top" align="left">58.49 &#x000B1; 0.91</td>
<td valign="top" align="left">71.28 &#x000B1; 0.69</td>
</tr> <tr>
<td valign="top" align="left">TADAM (Boris et al., <xref ref-type="bibr" rid="B3">2018</xref>)</td>
<td valign="top" align="left">58.50 &#x000B1; 0.30</td>
<td valign="top" align="left">76.70 &#x000B1; 0.30</td>
</tr> <tr>
<td valign="top" align="left">MTL (Sun et al., <xref ref-type="bibr" rid="B34">2019</xref>)</td>
<td valign="top" align="left">61.20 &#x000B1; 1.80</td>
<td valign="top" align="left">75.50 &#x000B1; 0.80</td>
</tr> <tr>
<td valign="top" align="left">DC (Lifchitz et al., <xref ref-type="bibr" rid="B22">2019</xref>)</td>
<td valign="top" align="left">62.53 &#x000B1; 0.19</td>
<td valign="top" align="left">78.95 &#x000B1; 0.13</td>
</tr> <tr>
<td valign="top" align="left">MetaOptNet (Lee et al., <xref ref-type="bibr" rid="B19">2019</xref>)</td>
<td valign="top" align="left">64.09 &#x000B1; 0.62</td>
<td valign="top" align="left">80.00 &#x000B1; 0.45</td>
</tr> <tr>
<td valign="top" align="left">DeepEMD (Zhang et al., <xref ref-type="bibr" rid="B42">2022</xref>)</td>
<td valign="top" align="left">65.91 &#x000B1; 0.82</td>
<td valign="top" align="left">82.41 &#x000B1; 0.56</td>
</tr> <tr>
<td valign="top" align="left">SIB (Hu et al., <xref ref-type="bibr" rid="B14">2020</xref>)</td>
<td valign="top" align="left">70.00 &#x000B1; 0.40</td>
<td valign="top" align="left">79.20 &#x000B1; 0.40</td>
</tr> <tr>
<td valign="top" align="left">BD-CSPN (Liu et al., <xref ref-type="bibr" rid="B23">2020</xref>)</td>
<td valign="top" align="left">70.31 &#x000B1; 0.93</td>
<td valign="top" align="left">81.89 &#x000B1; 0.60</td>
</tr> <tr>
<td valign="top" align="left">EPNet (Rodr&#x00301;&#x00131;guez et al., <xref ref-type="bibr" rid="B30">2020</xref>)</td>
<td valign="top" align="left">70.74 &#x000B1; 0.85</td>
<td valign="top" align="left">79.20 &#x000B1; 0.40</td>
</tr> <tr>
<td valign="top" align="left">Meta-BNNet (Gao et al., <xref ref-type="bibr" rid="B11">2023</xref>)</td>
<td valign="top" align="left">71.73 &#x000B1; 0.23</td>
<td valign="top" align="left">82.58 &#x000B1; 0.17</td>
</tr> <tr>
<td valign="top" align="left">CSMT</td>
<td valign="top" align="left">71.65 &#x000B1; 1.05</td>
<td valign="top" align="left">78.32 &#x000B1; 0.93</td>
</tr> <tr>
<td valign="top" align="left">FMPMIMC (without KL divergence)</td>
<td valign="top" align="left">72.05 &#x000B1; 0.67</td>
<td valign="top" align="left">81.20 &#x000B1; 0.80</td>
</tr> <tr>
<td valign="top" align="left">FMPMIMC (ours)</td>
<td valign="top" align="left"><bold>73.49</bold> &#x000B1; <bold>0.40</bold></td>
<td valign="top" align="left"><bold>83.55</bold> &#x000B1; <bold>0.75</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values are the best experimental results.</p>
</table-wrap-foot>
</table-wrap>



<p>As shown in <xref ref-type="table" rid="T3">Table 3</xref>, the FMPMIMC method achieves the highest average classification accuracy compared with the advanced baseline methods in the 5way-1shot and 5way-5shot experiments of the CIFAR-100 dataset, with an increase of 9.74 and 4.63%. This method has improved by 1.76 and 0.97% compared to advanced Meta-BNNet methods in experiments on the MiniImageNet dataset. Meanwhile, compared with the multiple model cooperation strategy of the FMPMIMC method and the single model CSMT method, the experimental results are improved, and the model is more stable, which shows the stability and effectiveness of the FMPMIMC method. For the KL divergence term in the FMPMIMC method, we conducted ablation experiments, and the experimental results in <xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref> show the effectiveness of the KL divergence term in the loss function. The experimental results of 5way-1shot on the two datasets are more prominent, which reflects the feature of the FMPMIMC method that enables the model to rapidly adapt to few-shot tasks.</p></sec>
<sec>
<title>4.4 The parameter setting experiment</title>
<p>In this subsection, we first analyze the relationship between the number of ensemble models and the effect of few-shot learning and the influence of interaction frequency of meta-training and meta-testing in the meta-optimization phase. Then, the relationship between the number of tasks in the candidate meta-task set and the few-shot learning effect is explored.</p>
<sec>
<title>4.4.1 The experiment on the number of ensemble model</title>
<p>In the 5way-1shot experiment of the CIFAR-100 dataset, we set the number of ensemble model to 1, 2, 3, 5, 10, and 20. Meanwhile, we set the interaction frequency of meta-training and meta-testing to 100, 200, 500, and 1,000 epochs. In <xref ref-type="fig" rid="F8">Figure 8</xref>, we compared the prediction accuracy using model cooperation strategy (existing KL divergence) and the average performance of multiple models(without KL divergence). The following are the analysis of the experimental results.</p>


<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>The classification accuracy of various numbers ensemble models. The solid line gives the classification accuracy after collaborative prediction of multiple models, and the average performance of a single model is plotted with a dashed line.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-g0008.tif"/>
</fig>


<p>As shown in <xref ref-type="fig" rid="F8">Figure 8</xref>, when a cooperative strategy is adopted for prediction, the accuracy of prediction improves with the increase in the number of cooperative models. When the number of models is 10, the prediction effect is the best, and the memory of multiple model ensemble is about 26 GB. When the number of models increases to 20, the performance of multiple model cooperation prediction is decreased. The performance of multiple model cooperation prediction is always better than its average performance, and its fluctuation is small, which reflects the effectiveness of the model cooperation strategy (existing KL divergence). Similarly, under the same number of ensemble models, the influence of different interaction frequencies between meta-training and meta-testing on model prediction results is explored. In many cases, the model gives the best prediction results when the interaction frequency is 500 epochs. The frequent interaction between meta-training and meta-testing will lead to overfitting of the model to few-shot data. Too little interaction frequency will lead to poor generalization of the model for few-shot tasks.</p></sec>
<sec>
<title>4.4.2 The experiment on the number of tasks in the candidate meta-task sets</title>
<p>In the 5way-1shot and 5way-5shot experiments of the CIFAR-100 dataset, we set the number of tasks in the candidate meta-task set to 2, 5, 10, and 15. In <xref ref-type="fig" rid="F9">Figure 9</xref>, we compare the influence of different numbers of tasks in the candidate meta-task set on classification accuracy and training process. The following are the analysis of the experimental results.</p>


<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>In the 5way-1shot experiment of the CIFAR-100 dataset, the effect of different numbers of tasks in the candidate meta-task set on the model classification accuracy.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-g0009.tif"/>
</fig>


<p>As shown in <xref ref-type="fig" rid="F9">Figure 9</xref>, the learning effect of the model is best when each class in the candidate meta-task sets corresponding to few-shot tasks contain 5 or 10 base class categories. When the number of tasks in the candidate meta-task sets is too few or too many, it will affect the learning effect of the model. When there are too few candidate meta-tasks, it will cause the model to overfit on few-shot tasks. When there are too many candidate meta-tasks, irrelevant meta-knowledge will be included, which will affect the learning effect of the model on few-shot tasks. Experiments show that the appropriate selection of meta-tasks from the base class is beneficial to few-shot learning.</p>
<p>As shown in <xref ref-type="fig" rid="F10">Figure 10</xref>, when the candidate meta-task set corresponding to each few-shot class is set to five base class categories, the training effect is the best. The model can adapt rapidly the few-shot classification tasks, and its average performance is higher than that of other cases.</p>
<fig id="F10" position="float">
<label>Figure 10</label>
<caption><p>The effect of different numbers of tasks in the candidate meta-task set on the model training process in the 5way-5shot experiment of the CIFAR-100 dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1391247-g0010.tif"/>
</fig></sec></sec>


<sec>
<title>4.5 Implementation details</title>
<p>We implement the FMPMIMC method based on PyTorch, which uses CNNs, Vgg16, and ResNet50 network architectures for different datasets. The network parameters are optimized by Adam optimizer. In the experiments conducted on the Omniglot dataset, setting the learning rate of the meta-learner to 10<sup>&#x02212;2</sup> yields the best effect, while setting the learning rate of meta-learner to 10<sup>&#x02212;3</sup> in experiments of the other two datasets produces the best effect. In the ensemble experiment of 20 models, the running memory exceeded 32 GB. We alleviate the problem of insufficient GPU running memory by reducing the size of the input batch of the network. All experiments are conducted on the NVIDIA Tesla V100 GPU to complete the training procedure.</p></sec></sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>Meta-learning methods based on optimization are widely used to improve the performance of few-shot learning. In this article, we provide a new idea for few-shot learning and study new methods for meta-task construction and multiple initial model cooperation. Considering the challenges discussed in our previous work, this article puts forward a full-phase meta-learning method based on situational meta-task construction for multiple model cooperation, which achieves few-shot learning and attempts to improve this problem. Experiments with both 5way-1shot and 5way-5shot tasks are conducted on several datasets, and the analyses prove the effectiveness of our proposed CSMT and FMPMIMC methods. Visualization experiments are more intuitive and vivid, which verifies that we provide useful meta-knowledge for few-shot tasks. The parameter setting experiments explore the influence of the iteration frequency between meta-training and meta-testing in the meta-optimization phase, the number of ensemble models, and the number of candidate meta-task set categories on training results. Our proposed methods are more suitable for providing relevant meta-knowledge to the model during the meta-training phase using novel few-shot tasks, which helps the model to learn and adapt. In future work, we will extend our methods to provide more generalized and useful meta-knowledge to the model during the meta-training period when the novel few-shot tasks are completely invisible.</p></sec>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p></sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>ZZ: Conceptualization, Methodology, Software, Writing&#x02014;original draft. LZ: Data curation, Funding acquisition, Investigation, Resources, Writing&#x02014;review &#x00026; editing. YW: Investigation, Software, Validation, Visualization, Writing&#x02014;review &#x00026; editing. NW: Conceptualization, Project administration, Supervision, Writing&#x02014;review &#x00026; editing.</p></sec>
</body>
<back>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was supported by the Basic Research Project (JCKY2021206B028).</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Afouras</surname> <given-names>T.</given-names></name> <name><surname>Chung</surname> <given-names>J. S.</given-names></name> <name><surname>Senior</surname> <given-names>A.</given-names></name> <name><surname>Vinyals</surname> <given-names>O.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>Deep audio-visual speech recognition</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>44</volume>, <fpage>8717</fpage>&#x02013;<lpage>8727</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2018.2889052</pub-id><pub-id pub-id-type="pmid">30582526</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bertinetto</surname> <given-names>L.</given-names></name> <name><surname>Torr</surname> <given-names>P.</given-names></name> <name><surname>Henriques</surname> <given-names>J.</given-names></name> <name><surname>Vedaldi</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Meta-learning with differentiable closed-form solvers,&#x0201D;</article-title> in <source>7th International Conference on Learning Representations, ICLR 2019</source> (<publisher-loc>New Orleans, LA</publisher-loc>).</citation>
</ref>
<ref id="B3">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Boris</surname> <given-names>N. O.</given-names></name> <name><surname>Pau</surname> <given-names>R.</given-names></name> <name><surname>Alexandre</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Tadam: task dependent adaptive metric for improved few-shot learning,&#x0201D;</article-title> in <source>32nd Conference on Neural Information Processing Systems (NIPS)</source> (<publisher-loc>Montreal, QC</publisher-loc>), <fpage>721</fpage>&#x02013;<lpage>731</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname> <given-names>L.</given-names></name></person-group> (<year>1996</year>). <article-title>Bagging predictors</article-title>. <source>Mach. Learn</source>. <volume>24</volume>, <fpage>123</fpage>&#x02013;<lpage>140</lpage>. <pub-id pub-id-type="doi">10.1007/BF00058655</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Fu</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Jiang</surname> <given-names>Y.</given-names></name> <name><surname>Xue</surname> <given-names>X.</given-names></name> <name><surname>Sigal</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>Multi-level semantic feature augmentation for one-shot learning</article-title>. <source>IEEE Transact. Image Process</source>. <volume>28</volume>, <fpage>4594</fpage>&#x02013;<lpage>4605</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2019.2910052</pub-id><pub-id pub-id-type="pmid">30969924</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elsken</surname> <given-names>T.</given-names></name> <name><surname>Staffler</surname> <given-names>B.</given-names></name> <name><surname>Metzen</surname> <given-names>J.</given-names></name> <name><surname>Hutter</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Meta-learning of neural architectures for few-shot learning,&#x0201D;</article-title> in <source>Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition</source>, <fpage>12362</fpage>&#x02013;<lpage>12372</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Finn</surname> <given-names>C.</given-names></name> <name><surname>Abbeel</surname> <given-names>P.</given-names></name> <name><surname>Levine</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Model-agnostic meta-learning for fast adaptation of deep networks,&#x0201D;</article-title> in <source>34th International Conference on Machine Learning, ICML</source> (<publisher-loc>Sydney, NSW</publisher-loc>), <fpage>1856</fpage>&#x02013;<lpage>1868</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fort</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <source>Gaussian Prototypical Networks for Few-Shot learning on Omniglot</source>. arXiv.1708.02735.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Freund</surname> <given-names>Y.</given-names></name> <name><surname>Schapire</surname> <given-names>R.</given-names></name></person-group> (<year>1996</year>). <article-title>Experiment with a new boosting algorithm</article-title>. <source>Morgan Kaufmann</source> <volume>96</volume>, <fpage>148</fpage>&#x02013;<lpage>156</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ganaie</surname> <given-names>M.</given-names></name> <name><surname>Hu</surname> <given-names>M.</given-names></name> <name><surname>Malik</surname> <given-names>A.</given-names></name> <name><surname>Tanveer</surname> <given-names>M.</given-names></name> <name><surname>Suganthan</surname> <given-names>P.</given-names></name></person-group> (<year>2022</year>). <article-title>Ensemble deep learning: a review</article-title>. <source>Eng. Appl. Artif. Intell</source>. <volume>115</volume>:<fpage>105151</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2022.105151</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>W.</given-names></name> <name><surname>Shao</surname> <given-names>M.</given-names></name> <name><surname>Shu</surname> <given-names>J.</given-names></name> <name><surname>Zhuang</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>Meta-bn net for few-shot learning</article-title>. <source>Front. Comp. Sci</source>. <volume>17</volume>, <fpage>131702</fpage>&#x02013;<lpage>131709</lpage>. <pub-id pub-id-type="doi">10.1007/s11704-021-1237-4</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Pu</surname> <given-names>N.</given-names></name> <name><surname>Lao</surname> <given-names>M.</given-names></name> <name><surname>Lew</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Few-shot and meta-learning methods for image understanding: a survey</article-title>. <source>Int. J. Multim. Inf</source> . 12. <pub-id pub-id-type="doi">10.1007/s13735-023-00279-4</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hospedales</surname> <given-names>T.</given-names></name> <name><surname>Antoniou</surname> <given-names>A.</given-names></name> <name><surname>Micaelli</surname> <given-names>P.</given-names></name> <name><surname>Storkey</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>Meta-learning in neural networks: a survey</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>44</volume>, <fpage>5149</fpage>&#x02013;<lpage>5169</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2021.3079209</pub-id><pub-id pub-id-type="pmid">33974543</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>S.</given-names></name> <name><surname>Moreno</surname> <given-names>P.</given-names></name> <name><surname>Xiao</surname> <given-names>Y.</given-names></name> <name><surname>Shen</surname> <given-names>X.</given-names></name> <name><surname>Obozinski</surname> <given-names>G.</given-names></name> <name><surname>Lawrence</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Empirical Bayes transductive meta-learning with synthetic gradients,&#x0201D;</article-title> in <source>8th International Conference on Learning Representations, ICLR 2020</source>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>H.</given-names></name> <name><surname>Diao</surname> <given-names>Z.</given-names></name> <name><surname>Shi</surname> <given-names>T.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>F.</given-names></name> <name><surname>Hu</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>A review of deep learning-based multiple-lesion recognition from medical images: classification, detection and segmentation</article-title>. <source>Comput. Biol. Med</source>. <volume>157</volume>:<fpage>106726</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106726</pub-id><pub-id pub-id-type="pmid">36924732</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kontschieder</surname> <given-names>P.</given-names></name> <name><surname>Fiterau</surname> <given-names>M.</given-names></name> <name><surname>Criminisi</surname> <given-names>A.</given-names></name> <name><surname>Bulo</surname> <given-names>S. R.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Deep neural decision forests,&#x0201D;</article-title> in <source>IEEE International Conference on Computer Vision</source> (<publisher-loc>Santiago</publisher-loc>), <fpage>1467</fpage>&#x02013;<lpage>1475</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lake</surname> <given-names>B.</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R.</given-names></name> <name><surname>Gross</surname> <given-names>J.</given-names></name> <name><surname>Tenenbaum</surname> <given-names>J.</given-names></name></person-group> (<year>2011</year>). <article-title>&#x0201C;One shot learning of simple visual concepts,&#x0201D;</article-title> in <source>Proceedings of the 33rd Annual Meeting of the Cognitive Science Society, CogSci</source> (<publisher-loc>Boston, MA</publisher-loc>), <fpage>2568</fpage>&#x02013;<lpage>2573</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>LeCun</surname> <given-names>Y.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>Nature</source> <volume>521</volume>, <fpage>436</fpage>&#x02013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id><pub-id pub-id-type="pmid">26017442</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Maji</surname> <given-names>S.</given-names></name> <name><surname>Ravichandran</surname> <given-names>A.</given-names></name> <name><surname>Soatto</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Meta-learning with differentiable convex optimization,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer vision and Pattern Recognition</source> (<publisher-loc>Long Beach, CA</publisher-loc>), <fpage>10657</fpage>&#x02013;<lpage>10665</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Sun</surname> <given-names>Z.</given-names></name> <name><surname>Xue</surname> <given-names>J.</given-names></name> <name><surname>Ma</surname> <given-names>Z.</given-names></name></person-group> (<year>2021</year>). <article-title>A concise review of recent few-shot meta-learning methods</article-title>. <source>Neurocomputing</source> <volume>456</volume>, <fpage>463</fpage>&#x02013;<lpage>468</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2020.05.114</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Zhou</surname> <given-names>F.</given-names></name> <name><surname>Chen</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2017</year>). <source>Meta-sgd: Learning to Learn Quickly for Few-Shot Learning</source>. arXiv.1707.09835.</citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lifchitz</surname> <given-names>Y.</given-names></name> <name><surname>Avrithis</surname> <given-names>Y.</given-names></name> <name><surname>Picard</surname> <given-names>S.</given-names></name> <name><surname>Bursuc</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Dense classification and implanting for few-shot learning,&#x0201D;</article-title> in <source>32nd IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Long Beach, CA</publisher-loc>), <fpage>9250</fpage>&#x02013;<lpage>9259</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Song</surname> <given-names>L.</given-names></name> <name><surname>Qin</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Prototype rectification for few-shot learning,&#x0201D;</article-title> in <source>Computer Vision- ECCV:16th European Conference</source> (<publisher-loc>Glasgow</publisher-loc>).<pub-id pub-id-type="pmid">38610574</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>W.</given-names></name> <name><surname>Lu</surname> <given-names>G.</given-names></name> <name><surname>Tian</surname> <given-names>Q.</given-names></name> <name><surname>Ling</surname> <given-names>N.</given-names></name></person-group> (<year>2022</year>). <article-title>Few-shot image classification: current status and research trends</article-title>. <source>Electronics</source> <volume>11</volume>, <fpage>1753</fpage>&#x02013;<lpage>1779</lpage>. <pub-id pub-id-type="doi">10.3390/electronics11111752</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>J.</given-names></name> <name><surname>Gong</surname> <given-names>P.</given-names></name> <name><surname>Ye</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>A survey on machine learning from few samples</article-title>. <source>Pattern Recognit</source>. <volume>139</volume>:<fpage>109480</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2023.109480</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Mishra</surname> <given-names>N.</given-names></name> <name><surname>Rohaninejad</surname> <given-names>M.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Abbeel</surname> <given-names>P.</given-names></name></person-group> (<year>2017</year>). <source>Meta-Learning With Temporal Convolutions</source>. arXiv.1707.03141.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nichol</surname> <given-names>A.</given-names></name> <name><surname>Achiam</surname> <given-names>J.</given-names></name> <name><surname>Schulman</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <source>On First-Order Meta-Learning Algorithms</source>. arXiv.1803.02999.<pub-id pub-id-type="pmid">33729779</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nikhil</surname> <given-names>M.</given-names></name> <name><surname>Mostafa</surname> <given-names>R.</given-names></name> <name><surname>Xi</surname> <given-names>C.</given-names></name> <name><surname>Pieter</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;A simple neural attentive meta-learner,&#x0201D;</article-title> in <source>6th International Conference on Learning Representations, ICLR 2018</source> - <italic>Conference Track Proceedings</italic> (Vancouver, BC).</citation>
</ref>
<ref id="B29">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rajeswaran</surname> <given-names>A.</given-names></name> <name><surname>Finn</surname> <given-names>C.</given-names></name> <name><surname>Kakade</surname> <given-names>S. M.</given-names></name> <name><surname>Levine</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Meta-learning with implicit gradients,&#x0201D;</article-title> in <source>33rd Conference on Neural Information Processing Systems (NIPS)</source> (<publisher-loc>Vancouver, BC</publisher-loc>).<pub-id pub-id-type="pmid">35124439</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rodr&#x000ED;guez</surname> <given-names>P.</given-names></name> <name><surname>Laradji</surname> <given-names>I.</given-names></name> <name><surname>Drouin</surname> <given-names>A.</given-names></name> <name><surname>Lacoste</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Embedding propagation: smoother manifold for few-shot classification,&#x0201D;</article-title> in <source>Computer Vision-ECCV: 16th European Conference</source> (<publisher-loc>Glasgow</publisher-loc>).</citation>
</ref>
<ref id="B31">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>T.</given-names></name> <name><surname>Zhou</surname> <given-names>T.</given-names></name> <name><surname>Long</surname> <given-names>G.</given-names></name> <name><surname>Jiang</surname> <given-names>J.</given-names></name> <name><surname>Pan</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Disan: directional self-attention network for rnn/cnn-free language understanding,&#x0201D;</article-title> in <source>32nd AAAI Conference on Artificial Intelligence, AAAI</source> (<publisher-loc>New Orleans, LA</publisher-loc>), <fpage>5446</fpage>&#x02013;<lpage>5455</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Snell</surname> <given-names>J.</given-names></name> <name><surname>Swersky</surname> <given-names>K.</given-names></name> <name><surname>Zemel</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>Prototypical networks for few-shot learning</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>30</volume>, <fpage>4078</fpage>&#x02013;<lpage>4088</lpage>. arXiv.1703.05175.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Cai</surname> <given-names>P.</given-names></name> <name><surname>Mondal</surname> <given-names>S.</given-names></name> <name><surname>Sahoo</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>A comprehensive survey of few-shot learning: evolution, applications, challenges, and opportunities. <italic>ACM Comp</italic></article-title>. <source>Surv</source>. <volume>55</volume>:<fpage>3582688</fpage>. <pub-id pub-id-type="doi">10.1145/3582688</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>Q.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Chua</surname> <given-names>T.</given-names></name> <name><surname>Schiele</surname> <given-names>B.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Meta-transfer learning for few-shot learning,&#x0201D;</article-title> in <source>32nd IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Long Beach, CA</publisher-loc>), <fpage>403</fpage>&#x02013;<lpage>412</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vanschoren</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <source>Meta-learning: A Survey.</source> arXiv.1810.03548.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vettoruzzo</surname> <given-names>A.M.R B</given-names></name> <name><surname>Vanschoren</surname> <given-names>J.</given-names></name> <name><surname>Rognvaldsson</surname> <given-names>T.</given-names></name> <name><surname>Santosh</surname> <given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>Advances and challenges in meta-learning: a technical review</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <pub-id pub-id-type="doi">10.1109/TPAMI.2024.3357847</pub-id><pub-id pub-id-type="pmid">38265905</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vinyals</surname> <given-names>O.</given-names></name> <name><surname>Blundell</surname> <given-names>C.T</given-names></name> <name><surname>Lillicrap Wierstra</surname> <given-names>D.</given-names></name></person-group> (<year>2016</year>). <article-title>Matching networks for one shot learning</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>0</volume>, <fpage>3637</fpage>&#x02013;<lpage>3645</lpage>. arXiv.1606.04080.</citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Yao</surname> <given-names>Q.</given-names></name> <name><surname>Kwok</surname> <given-names>J. T.</given-names></name> <name><surname>Ni</surname> <given-names>L.</given-names></name></person-group> (<year>2020</year>). <article-title>Generalizing from a few examples: a survey on few-shot learning</article-title>. <source>ACM Comp. Surv</source>. <volume>53</volume>, <fpage>1</fpage>&#x02013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1145/3386252</pub-id><pub-id pub-id-type="pmid">37067965</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wolpert</surname> <given-names>D. H.</given-names></name></person-group> (<year>1992</year>). <article-title>Stacked generalization</article-title>. <source>Neur. Netw</source>. <volume>5</volume>, <fpage>241</fpage>&#x02013;<lpage>259</lpage>. <pub-id pub-id-type="doi">10.1016/S0893-6080(05)80023-1</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>J.</given-names></name> <name><surname>Tan</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Rui</surname> <given-names>Y.</given-names></name> <name><surname>Tao</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>Hierarchical deep click feature prediction for fine-grained image recognition</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>44</volume>, <fpage>563</fpage>&#x02013;<lpage>578</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2019.2932058</pub-id><pub-id pub-id-type="pmid">31380745</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>W.</given-names></name> <name><surname>Xiao</surname> <given-names>Z.</given-names></name></person-group> (<year>2024</year>). <article-title>Few-shot learning based on deep learning: a survey</article-title>. <source>Math. Biosci. Eng</source>. <volume>21</volume>:<fpage>2024029</fpage>. <pub-id pub-id-type="doi">10.3934/mbe.2024029</pub-id><pub-id pub-id-type="pmid">38303439</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>C.</given-names></name> <name><surname>Cai</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>G.</given-names></name> <name><surname>Shen</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>Deepemd: differentiable earth mover&#x00027;s distance for few-shot learning</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>45</volume>, <fpage>5632</fpage>&#x02013;<lpage>5648</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2022.3217373</pub-id><pub-id pub-id-type="pmid">36288227</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Zhao</surname> <given-names>X.</given-names></name> <name><surname>Tian</surname> <given-names>Q.</given-names></name></person-group> (<year>2019</year>). <article-title>Spontaneous speech emotion recognition using multiscale deep convolutional lstm</article-title>. <source>IEEE Transact. Affect. Comp</source>. <volume>13</volume>, <fpage>680</fpage>&#x02013;<lpage>688</lpage>. <pub-id pub-id-type="doi">10.1109/TAFFC.2019.2947464</pub-id></citation>
</ref>
<ref id="B44">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>F.</given-names></name> <name><surname>Wu</surname> <given-names>B.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name></person-group> (<year>2018</year>). <source>Deep Meta-Learning: Learning to Learn in the Concept Space</source>. arXiv.1802.03596.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>L.</given-names></name> <name><surname>Cui</surname> <given-names>P.</given-names></name> <name><surname>Jia</surname> <given-names>X.</given-names></name> <name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Tian</surname> <given-names>Q.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Learning to select base classes for few-shot classification,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, <fpage>4624</fpage>&#x02013;<lpage>4633</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Z.</given-names></name> <name><surname>Feng</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Deep forest: towards an alternative to deep neural networks,&#x0201D;</article-title> in <source>26th International Joint Conference on Artificial Intelligence (IJCAI)</source> (<publisher-loc>Melbourne, VIC</publisher-loc>) 3553&#x02013;3559.</citation>
</ref>
</ref-list>
</back>
</article>