<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Physiol.</journal-id>
<journal-title>Frontiers in Physiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Physiol.</abbrev-journal-title>
<issn pub-type="epub">1664-042X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1337554</article-id>
<article-id pub-id-type="doi">10.3389/fphys.2024.1337554</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A novel dilated contextual attention module for breast cancer mitosis cell detection</article-title>
<alt-title alt-title-type="left-running-head">Li et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphys.2024.1337554">10.3389/fphys.2024.1337554</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Li</surname>
<given-names>Zhiqiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2625524/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Li</surname>
<given-names>Xiangkui</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2192019/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Weixuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lyu</surname>
<given-names>He</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2358289/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tang</surname>
<given-names>Xuezhi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhou</surname>
<given-names>Chenchen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2345676/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Fanxin</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Luo</surname>
<given-names>Bin</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Jiang</surname>
<given-names>Yulian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2575195/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Xingwen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xiang</surname>
<given-names>Wei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2358015/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Key Laboratory of Electronic and Information Engineering</institution>, <institution>State Ethnic Affairs Commission</institution>, <institution>Southwest Minzu University</institution>, <addr-line>Chengdu</addr-line>, <addr-line>Sichuan</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Computer Science and Technology</institution>, <institution>Harbin University of Science and Technology</institution>, <addr-line>Harbin</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Chongqing Key Laboratory of Computational Intelligence</institution>, <institution>Chongqing University of Posts and Telecommunications</institution>, <addr-line>Chongqing</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Sichuan Huhui Software Co., LTD.</institution>, <addr-line>Mianyang</addr-line>, <addr-line>Sichuan</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/29651/overview">Zhihui Wang</ext-link>, Houston Methodist Research Institute, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/987955/overview">Massimo Salvi</ext-link>, Polytechnic University of Turin, Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1806665/overview">Jan Kubicek</ext-link>, VSB-Technical University of Ostrava, Czechia</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yulian Jiang, <email>jyl-ee@swun.edu.cn</email>
</corresp>
<fn fn-type="equal" id="fn001">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors share first authorship</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>25</day>
<month>01</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1337554</elocation-id>
<history>
<date date-type="received">
<day>14</day>
<month>11</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>03</day>
<month>01</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Li, Li, Wu, Lyu, Tang, Zhou, Xu, Luo, Jiang, Liu and Xiang.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Li, Li, Wu, Lyu, Tang, Zhou, Xu, Luo, Jiang, Liu and Xiang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Background and object:</bold> Mitotic count (MC) is a critical histological parameter for accurately assessing the degree of invasiveness in breast cancer, holding significant clinical value for cancer treatment and prognosis. However, accurately identifying mitotic cells poses a challenge due to their morphological and size diversity.</p>
<p>
<bold>Objective:</bold> We propose a novel end-to-end deep-learning method for identifying mitotic cells in breast cancer pathological images, with the aim of enhancing the performance of recognizing mitotic cells.</p>
<p>
<bold>Methods:</bold> We introduced the Dilated Cascading Network (DilCasNet) composed of detection and classification stages. To enhance the model&#x2019;s ability to capture distant feature dependencies in mitotic cells, we devised a novel Dilated Contextual Attention Module (DiCoA) that utilizes sparse global attention during the detection. For reclassifying mitotic cell areas localized in the detection stage, we integrate the EfficientNet-B7 and VGG16 pre-trained models (InPreMo) in the classification step.</p>
<p>
<bold>Results:</bold> Based on the canine mammary carcinoma (CMC) mitosis dataset, DilCasNet demonstrates superior overall performance compared to the benchmark model. The specific metrics of the model&#x2019;s performance are as follows: F1 score of 82.9%, Precision of 82.6%, and Recall of 83.2%. With the incorporation of the DiCoA attention module, the model exhibited an improvement of over 3.5% in the F1 during the detection stage.</p>
<p>
<bold>Conclusion:</bold> The DilCasNet achieved a favorable detection performance of mitotic cells in breast cancer and provides a solution for detecting mitotic cells in pathological images of other cancers.</p>
</abstract>
<kwd-group>
<kwd>mitosis detection</kwd>
<kwd>mitotic count</kwd>
<kwd>dilated attention</kwd>
<kwd>whole slide image</kwd>
<kwd>multi-stage deep learning</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Physiology and Medicine</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Breast cancer is among the most common malignancies, with a high incidence and mortality rate among women worldwide (<xref ref-type="bibr" rid="B51">Xu et al., 2023</xref>). Histopathological image analysis has long been regarded as the &#x201c;gold standard&#x201d; in cancer diagnosis and prognosis evaluation (<xref ref-type="bibr" rid="B14">Gurcan et al., 2009</xref>). The identification of molecular quantities and features within patients&#x2019; tumors is crucial for the clinical treatment and prognosis assessment of cancer patients (<xref ref-type="bibr" rid="B12">Dai et al., 2023</xref>). Within histopathological image analysis, the mitotic count is recognized as a critical histological parameter for diagnosing and grading cancer (<xref ref-type="bibr" rid="B11">Cree et al., 2021</xref>). However, the current MC still relies on manual counting of mitotic figures through an optical microscope (<xref ref-type="bibr" rid="B5">Bertram et al., 2020</xref>), and even pathologists only maintain moderate consistency in identifying mitotic cells (<xref ref-type="bibr" rid="B21">Ibrahim et al., 2021</xref>).</p>
<p>To achieve automated mitotic detection to assist pathologists in diagnosis, traditional machine learning approaches depend on prior knowledge, employing carefully designed handcraft feature extractors to process features and integrate various machine learning classifiers for mitotic cell identification (<xref ref-type="bibr" rid="B29">Lu and Mandal, 2014</xref>; <xref ref-type="bibr" rid="B37">Paul and Mukherjee, 2015</xref>; <xref ref-type="bibr" rid="B32">Mathew et al., 2021</xref>). Although manual feature extraction contributes to the comprehension of mitotic cell characteristics, their generalization performance across large-scale datasets is constrained.</p>
<p>With the continuous advancement of deep learning, convolutional neural networks (CNNs) have provided new solutions for mitotic cell detection (<xref ref-type="bibr" rid="B25">Lecun et al., 2015</xref>; <xref ref-type="bibr" rid="B27">Lin et al., 2016</xref>; <xref ref-type="bibr" rid="B18">Huang et al., 2017</xref>). Concurrently, the availability of publicly accessible datasets featuring expert-annotated images of mitotic cells, such as ICPR MITOS-2012 (<xref ref-type="bibr" rid="B30">Ludovic et al., 2013</xref>), AMIDA 2013 (<xref ref-type="bibr" rid="B46">Veta et al., 2015</xref>), ICPR MITOS-ATYPIA-2014 (<xref ref-type="bibr" rid="B34">MITOS-ATYPIA-14., 2014</xref>), and TUPAC 2016 (<xref ref-type="bibr" rid="B45">Veta et al., 2019</xref>), has facilitated the application of deep learning methods in mitotic cell detection. However, these datasets only contain annotated mitotic images corresponding to High Power Fields (HPF) (<xref ref-type="bibr" rid="B5">Bertram et al., 2020</xref>) in hotspots and lack annotations for most areas of whole slide images (WSI). Recently, two extensive WSI datasets with annotated mitotic cells have been introduced: the canine cutaneous mast cell tumor (CCMCT) dataset (<xref ref-type="bibr" rid="B6">Bertram et al., 2019</xref>) and the canine mammary carcinoma (CMC) dataset (<xref ref-type="bibr" rid="B3">Aubreville et al., 2020</xref>). These datasets enable automatic mitotic detection models to learn from a more extensive collection of mitotic images and their contextual information (<xref ref-type="bibr" rid="B6">Bertram et al., 2019</xref>).</p>
<p>Previous studies have directly applied deep learning models for the recognition of mitotic cells (<xref ref-type="bibr" rid="B10">Cire&#x15f;an et al., 2013</xref>; <xref ref-type="bibr" rid="B52">Zerhouni et al., 2017</xref>), but these existing methods lack adequate domain adaptability. Currently, mitotic recognition methods typically utilize multi-stage models that integrate various tasks including detection, segmentation, and classification (<xref ref-type="bibr" rid="B26">Li et al., 2018</xref>; <xref ref-type="bibr" rid="B2">Alom et al., 2020</xref>), which perform better than single models. The diverse and intricate morphological features of mitotic cells across different cell cycle phases result in significant heterogeneity. Moreover, mitotic cells are often sparsely distributed and can be easily mistaken for other cell types, such as apoptotic cells and densely packed nuclear cells, when compared to normal cells (<xref ref-type="bibr" rid="B20">Ibrahim et al., 2022</xref>). Currently, multi-stage mitotic detection and classification models have not specifically focused on the impact of feature extraction and application on model performance.</p>
<p>We propose a two-stage Dilated Cascading Network (DilCasNet) to improve the performance of mitosis detection. In the mitosis cell detection stage, inspired by the Extended Contextual Attention (DiNA) (<xref ref-type="bibr" rid="B15">Hassani and Shi, 2022</xref>) and Polar Attention Network (PolarNet) (<xref ref-type="bibr" rid="B48">Wei et al., 2022</xref>), we propose a novel attention module, namely, Dilated Contextual Attention (DiCoA), and combine it with the Feature Pyramid Network (FPN) (<xref ref-type="bibr" rid="B27">Lin et al., 2016</xref>) of the Cascade RCNN (<xref ref-type="bibr" rid="B8">Cai and Vasconcelos, 2017</xref>) detection network to enhance the detection performance of mitosis cells. In the classification stage, we integrate the EfficientNet-B7 (<xref ref-type="bibr" rid="B42">Tan and Le, 2019</xref>) and VGG16 (<xref ref-type="bibr" rid="B41">Simonyan and Zisserman, 2015</xref>) pre-trained models to enhance the model&#x2019;s classification performance. The main contributions of this study are as follows:<list list-type="simple">
<list-item>
<p>(1) Introducing DiCoA, a sparse global attention module based on the self-attention mechanism, which achieves a larger receptive field by sparsifying keys and values, enabling the model to benefit in the challenging task of mitotic cell detection with complex morphologies. Experimental evidence demonstrates that incorporating DiCoA into the FPN of the Cascade R-CNN detection network reduces false positive predictions and enhances the model&#x2019;s performance in recognizing mitotic cells.</p>
</list-item>
<list-item>
<p>(2) To enhance the feature extraction of mitotic cells by the classification model, we integrate the EfficientNet-B7 and VGG16 pre-trained models (InPreMo), further improving the performance of the mitotic cell detection model by combining various CNN pre-trained models.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>Many automated algorithms for mitosis cell detection have been proposed to assist pathologists in diagnosis. In the early stages, manual design and feature selection methods were typically employed to achieve automated detection (<xref ref-type="bibr" rid="B32">Mathew et al., 2021</xref>). The entire process is generally divided into two steps: First, restrict the detection scope to specific candidate regions selected for segmentation. Subsequently, directly extract features from the image, including texture, statistical, and morphological features (<xref ref-type="bibr" rid="B24">Irshad et al., 2013</xref>; <xref ref-type="bibr" rid="B37">Paul and Mukherjee, 2015</xref>; <xref ref-type="bibr" rid="B35">Nateghi et al., 2017</xref>), or extract features from different color spaces (<xref ref-type="bibr" rid="B23">Irshad et al., 2014b</xref>; <xref ref-type="bibr" rid="B22">2014a</xref>; <xref ref-type="bibr" rid="B29">Lu and Mandal, 2014</xref>). The extracted features are then used to develop decision trees, random forests (RF), support vector machines (SVM) (<xref ref-type="bibr" rid="B44">Udousoro, 2020</xref>), and other classifiers to distinguish non-mitotic cells from mitotic cells in pathological slides. These methods have demonstrated competitive performance on datasets such as ICPR MITOS-2012, AMIDA 2013, and ICPR MITOS-ATYPIA-2014. However, manual feature extraction primarily relies on handcrafted feature extractors, making the process labor-intensive and challenging to extract deep abstract features.</p>
<p>With the development of deep learning, CNNs have demonstrated excellent capabilities in automatic feature extraction and learning and have achieved significant performance in tasks such as image classification, object detection, and semantic segmentation (<xref ref-type="bibr" rid="B25">Lecun et al., 2015</xref>; <xref ref-type="bibr" rid="B39">Ren et al., 2015</xref>; <xref ref-type="bibr" rid="B49">Weng and Zhu, 2015</xref>). Consequently, CNNs have found widespread applications in medical image processing (<xref ref-type="bibr" rid="B43">Tran et al., 2021</xref>). In the mitosis detection research, some experts and scholars choose the recently popular deep convolutional neural networks for automatic mitosis detection. Employing deep learning algorithms, pixel-wise classifiers have been developed to compute the probability of each pixel being associated with a mitotic event (<xref ref-type="bibr" rid="B10">Cire&#x15f;an et al., 2013</xref>; <xref ref-type="bibr" rid="B52">Zerhouni et al., 2017</xref>). These approaches demonstrate a high level of accuracy. To further augment the model&#x2019;s capacity for extracting mitotic cell features, multi-stage deep learning approaches (<xref ref-type="bibr" rid="B26">Li et al., 2018</xref>; <xref ref-type="bibr" rid="B2">Alom et al., 2020</xref>) were adopted, combining detection, segmentation, and classification tasks to develop two-stage and three-stage models. Similar to existing studies, our approach also adopts a two-stage method for mitotic classification detection. The utilization of multiple classifiers (<xref ref-type="bibr" rid="B47">Wang et al., 2014</xref>; <xref ref-type="bibr" rid="B4">Beevi et al., 2017</xref>; <xref ref-type="bibr" rid="B31">Mahmood et al., 2020</xref>), combined with handcrafted features, segmentation, or detection methods, achieved mitosis detection in a cascaded manner, further strengthening the model&#x2019;s capability in feature extraction and mitotic cell recognition. These methods have all demonstrated various degrees of performance improvement on the ICPR MITOS-2012 and MITOS-ATYPIA-2014 datasets. However, these two datasets have limited images and samples, and most of the WSI regions lack images and annotations, which poses challenges for model training. In accordance with recommendations from existing studies (<xref ref-type="bibr" rid="B3">Aubreville et al., 2020</xref>), we utilized a larger-scale CMC dataset for model training and evaluation.</p>
<p>Due to the diverse shapes of mitotic cells, attention modules are widely considered effective for better feature extraction from data (<xref ref-type="bibr" rid="B7">Brauwers and Frasincar, 2022</xref>). <xref ref-type="bibr" rid="B17">Hu et al. (2020)</xref> introduced Squeeze-and-Excitation Networks (SENet), which construct interdependencies among feature channels through weighted operations to enhance model expressiveness. Regarding spatial information processing, <xref ref-type="bibr" rid="B19">Huang et al. (2018)</xref> proposed the Criss-Cross Network (CCNet) to help the network obtain contextual information from the image, allowing each pixel to perceive its relevance to the entire image. To simultaneously focus on channel and spatial information, <xref ref-type="bibr" rid="B50">Woo et al. (2018)</xref> introduced the Convolutional Block Attention Module (CBAM), which combines channel and spatial attention, maintaining a small overhead while improving the model&#x2019;s focus on spatial and channel features. Multiple studies have demonstrated that introducing attention modules effectively enhances the model&#x2019;s feature extraction capability. However, these classical attention mechanisms are not specifically designed for mitotic detection and cannot fully leverage the potential of attention mechanisms to enhance model performance in mitotic classification detection. Therefore, we have devised a novel attention mechanism to address this purpose. Simultaneously, transfer learning methods (<xref ref-type="bibr" rid="B36">Pan and Yang, 2010</xref>) have been widely applied in various tasks to alleviate the issues of training network models requiring time and limited training data, which are of great significance for automated mitotic cell detection research. These methods have positive implications for enhancing the performance of automated detection of mitotic cells.</p>
</sec>
<sec sec-type="materials|methods" id="s3">
<title>3 Materials and methods</title>
<sec id="s3-1">
<title>3.1 Materials</title>
<sec id="s3-1-1">
<title>3.1.1 CMC dataset</title>
<p>This study utilized a dataset of 21 WSIs for CMC (<xref ref-type="bibr" rid="B3">Aubreville et al., 2020</xref>), which encompassed three different modes of annotations: Manually Expert Labeled (MEL), Object-Detection Augmented and Expert Labeled (ODAEL), and Clustering and Object-Detection Augmented and Expert Labeled (CODAEL). To facilitate comparison with prior research (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>), we followed the methodology presented in the previous study, using the CODAEL annotations for training and testing, with 14 of the WSIs in the training set and the remaining 7 in the test set. Detailed information on this dataset is provided in <xref ref-type="sec" rid="s12">Supplementary Data SA.1.1</xref>.</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 CCMCT dataset</title>
<p>This study conducts generalization validation using the CCMCT dataset, which comprises 32 WSIs. The dataset includes three different annotation methods for various categories: Manually Expert Labeled (MEL), Hard-Example Augmented Expert Labelled (HEAEL), and Object-Detection Augmented Expert Labelled (ODAEL). To facilitate comparison with prior research (<xref ref-type="bibr" rid="B6">Bertram et al., 2019</xref>), we performed testing on a test set containing 11 WSIs. For detailed information on this dataset, please refer to <xref ref-type="sec" rid="s12">Supplementary Data SA.1.2</xref>.</p>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Methods</title>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> illustrates the overall workflow of the mitosis detection model DilCasNet. Large WSIs, after undergoing preprocessing steps such as cropping, are input into the model to detect mitotic cells. The model is primarily divided into two stages: the mitotic detection stage utilizing Cascade R-CNN with DiCoA attention and the mitotic cell classification stage incorporating pre-trained models, EfficientNet-B7 and VGG16.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Overall Study Design. At the top, the flowchart of our method is presented; the middle section provides detailed implementation steps; after preprocessing of the massive WSIs, different-sized images for the detection and classification stages of model training are generated; on the right, our method&#x2019;s performance is demonstrated in various aspects; at the bottom, a brief overview of how our method processes WSI is outlined.</p>
</caption>
<graphic xlink:href="fphys-15-1337554-g001.tif"/>
</fig>
<sec id="s3-2-1">
<title>3.2.1 DiCoA module</title>
<p>The design of DiCoA is illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>. In the first step, DiCoA obtains the attention score matrix of the dilation contextual by calculating the self-attention within the neighborhood of the feature expansion interval, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref> (I). Subsequently, DiCoA further generates new feature maps by weighting the attention scores in different directions within the interval neighborhood region, as depicted in <xref ref-type="fig" rid="F2">Figure 2</xref> (II).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The structural diagram of DiCoA, where <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represent the number of channels, height, and width of the input feature map, respectively.</p>
</caption>
<graphic xlink:href="fphys-15-1337554-g002.tif"/>
</fig>
<p>Calculating the attention scores for the dilated contextual of feature maps: Given the input feature map <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, projections of the feature map queries (<inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>), keys (<inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>), and values (<inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>) are obtained through 2D convolution. The formula for the matrix of attention scores for the dilated contextual, <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, as shown in Formula <xref ref-type="disp-formula" rid="e1">1</xref>.<disp-formula id="e1">
<mml:math id="m9">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2299;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>&#x3b4;</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denote the coordinates of the <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th row and <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th column in the feature map. <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the nearest neighbor query for the pixel at coordinates <inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:mo>&#x2299;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the dot product operation. <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the total number of pixels in a neighborhood of size <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>; we set the <inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to 3 and the <inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to 9. <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the dilation value; we set the <inline-formula id="inf21">
<mml:math id="m22">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to 1. <inline-formula id="inf22">
<mml:math id="m23">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>&#x3b4;</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the positions of neighbors in the <inline-formula id="inf23">
<mml:math id="m24">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-dilated neighborhood of the <inline-formula id="inf24">
<mml:math id="m25">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>-th pixel (see <xref ref-type="sec" rid="s12">Supplementary Data SA.2.1</xref>). <inline-formula id="inf25">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>&#x3b4;</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the key for the <inline-formula id="inf26">
<mml:math id="m27">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>-th pixel in the <inline-formula id="inf27">
<mml:math id="m28">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-dilated neighborhood of size <inline-formula id="inf28">
<mml:math id="m29">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. In addition, <xref ref-type="sec" rid="s12">Supplementary Data SA.2.2</xref> provides the update method for attention scores in the network.</p>
<p>Updating the feature maps of the network: The dot product operation between <inline-formula id="inf30">
<mml:math id="m31">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf31">
<mml:math id="m32">
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> results an dilated contextual attention score matrix, <inline-formula id="inf32">
<mml:math id="m33">
<mml:mrow>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, of size <inline-formula id="inf33">
<mml:math id="m34">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. After resizing, it becomes a <inline-formula id="inf34">
<mml:math id="m35">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> matrix. Finally, matrix operation between <inline-formula id="inf35">
<mml:math id="m36">
<mml:mrow>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf36">
<mml:math id="m37">
<mml:mrow>
<mml:mi mathvariant="bold-italic">V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> results in the weighted feature map <inline-formula id="inf37">
<mml:math id="m38">
<mml:mrow>
<mml:mi mathvariant="bold-italic">y</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> with dimensions <inline-formula id="inf38">
<mml:math id="m39">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, as depicted in Formula <xref ref-type="disp-formula" rid="e2">2</xref>.<disp-formula id="e2">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">y</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>D</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:msubsup>
<mml:mo>&#xd7;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf39">
<mml:math id="m41">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the channel size of the feature map. <inline-formula id="inf40">
<mml:math id="m42">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the data normalization method. <inline-formula id="inf41">
<mml:math id="m43">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents the value projection of the <inline-formula id="inf42">
<mml:math id="m44">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-dilated neighborhood of size <inline-formula id="inf43">
<mml:math id="m45">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> for the <inline-formula id="inf44">
<mml:math id="m46">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>-th pixel, which can be expressed by Formula <xref ref-type="disp-formula" rid="e3">3</xref>.<disp-formula id="e3">
<mml:math id="m47">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>&#x3b4;</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mtd>
<mml:mtd>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>&#x3b4;</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mtd>
<mml:mtd>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>&#x3b4;</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Network architecture</title>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> shows the structure of our DilCasNet, which comprises two stages: detection and classification. In the detection stage, we employ the Cascade R-CNN object detection network and introduce the DiCoA attention module to predict the positions of mitotic figures in WSI. Subsequently, a window relocation algorithm is applied to reassess low-quality false-positive predictions around the image borders, as illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref> (I). In the classification stage, we refine the detected targets by center adjustment to better align with the image center. We then incorporate EfficientNet-B7 and VGG16 pre-trained models to reevaluate the confidence scores for each image&#x2019;s targets, resulting in the final predictions, as depicted in <xref ref-type="fig" rid="F3">Figure 3</xref> (II).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Mitosis detection model overall architecture diagram. I. Detection stage; II. Classification stage. Where <inline-formula id="inf45">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf46">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf47">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and represent the channels, height, and width of the feature map, respectively.</p>
</caption>
<graphic xlink:href="fphys-15-1337554-g003.tif"/>
</fig>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Detection stage</title>
<p>In the detection stage, we employed the Cascade R-CNN object detection network with an input image size of <inline-formula id="inf48">
<mml:math id="m51">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>512</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>512</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The network outputs a set of bounding boxes {<inline-formula id="inf49">
<mml:math id="m52">
<mml:mrow>
<mml:mfenced open="" close="}" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="italic">det</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf50">
<mml:math id="m53">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> represents the center coordinates of the predicted target, <inline-formula id="inf51">
<mml:math id="m54">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf52">
<mml:math id="m55">
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denote the width and height of the bounding box, and <inline-formula id="inf53">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="italic">det</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> indicates the confidence of a positive target.</p>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> (I) illustrates that our model utilizes ResNet-101 (<xref ref-type="bibr" rid="B16">He et al., 2016</xref>) as the backbone network to extract features. These features are fed into a Feature Pyramid Network (FPN) layer enhanced with DiCoA to integrate multi-scale feature information. The DiCoA module is incorporated during the bottom-up process of the FPN module, positioned after the <inline-formula id="inf54">
<mml:math id="m57">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution layer of the C5 layer. This adjustment allows the feature extraction regions to adapt to the actual size of mitotic cells, as detailed in <xref ref-type="sec" rid="s12">Supplementary Data SA.3</xref>.</p>
<p>After undergoing DiCoA processing, the feature map will yield weighted feature map <inline-formula id="inf55">
<mml:math id="m58">
<mml:mrow>
<mml:mi mathvariant="bold-italic">y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and dilated contextual attention scores <inline-formula id="inf56">
<mml:math id="m59">
<mml:mrow>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The feature map <inline-formula id="inf57">
<mml:math id="m60">
<mml:mrow>
<mml:mi mathvariant="bold-italic">y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is input into the Top-down process of the FPN module at the P5 layer to facilitate the transmission of high-level semantic information to lower-level feature maps. Simultaneously, <inline-formula id="inf58">
<mml:math id="m61">
<mml:mrow>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is transformed into a scalar result <inline-formula id="inf59">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>The multi-scale feature maps fused by the FPN are input into the Region Proposal Network (RPN) (<xref ref-type="bibr" rid="B39">Ren et al., 2015</xref>) for generating candidate regions of mitotic cells. Subsequently, these target candidate regions are passed through the Cascade ROI Head, which consists of a series of cascaded classification heads and regression heads. This process yields the regression parameters <inline-formula id="inf60">
<mml:math id="m63">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> for the bounding box of the target, along with the original target confidence score <inline-formula id="inf61">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, the original target confidence <inline-formula id="inf62">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the attention score <inline-formula id="inf63">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from DiCoA are weighted by the factor <inline-formula id="inf64">
<mml:math id="m67">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, resulting in the ultimate confidence <inline-formula id="inf65">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="italic">det</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> or the mitotic cell bounding box. The update is formulated as follows in Eq. <xref ref-type="disp-formula" rid="e4">4</xref>:<disp-formula id="e4">
<mml:math id="m69">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="italic">det</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>In this context, where <inline-formula id="inf66">
<mml:math id="m70">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the confidence allocation weight for DiCoA, we set the weight (<inline-formula id="inf67">
<mml:math id="m71">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) to 0.5 (refer to <xref ref-type="sec" rid="s12">Supplementary Data SB.5</xref>).</p>
<p>To maintain model performance stability during data sampling, we opt not to employ dynamic queries. Instead, we sample from the WSI and then train and test. Simultaneously, within the DiCoA module, we set the channel number of the 2D convolution to 256, the kernel size to 1, and the stride to 1. The neighborhood size (<inline-formula id="inf68">
<mml:math id="m72">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is set to 3 (refer to <xref ref-type="sec" rid="s12">Supplementary Data SB.2</xref>), resulting in <inline-formula id="inf69">
<mml:math id="m73">
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> being 9. The dilation value (<inline-formula id="inf70">
<mml:math id="m74">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is set to 1. The configuration for <inline-formula id="inf71">
<mml:math id="m75">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>&#x3b4;</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is provided in <xref ref-type="sec" rid="s12">Supplementary Data SA.2.1</xref>.</p>
<p>Following the detection stage, we applied a window relocation method (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>) to eliminate low-quality predictions around the borders of the sliding window frames.</p>
<p>The detailed structural parameters of the detection stage can be found in <xref ref-type="sec" rid="s12">Supplementary Data SA.6.1</xref>.</p>
</sec>
<sec id="s3-2-4">
<title>3.2.4 Classification stage</title>
<p>The classification stage occurs after the target detection stage, as illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref> (II). Initially, we employ the target center adjustment method (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>), which adjusts the extracted target center coordinates <inline-formula id="inf72">
<mml:math id="m76">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> to the image center <inline-formula id="inf73">
<mml:math id="m77">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, with specific update details outlined in <xref ref-type="sec" rid="s12">Supplementary Data SA.4</xref>. Subsequently, the EfficientNet-B7 and VGG16 pre-trained models receive input images of size <inline-formula id="inf74">
<mml:math id="m78">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>128</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>128</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and generate feature map outputs with consistent width and height, denoted as <inline-formula id="inf75">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf76">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, respectively. Here, <inline-formula id="inf77">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf78">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the channel numbers of the feature maps from different pre-trained models. These two feature maps are concatenated along the channel dimension to form a feature map of size <inline-formula id="inf79">
<mml:math id="m83">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Following this, a 2D global average pooling operation is applied to compress the concatenated feature map to dimensions <inline-formula id="inf80">
<mml:math id="m84">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and then passed through a fully connected layer to output the target confidence. The final output result is utilized to update the results from the detection stage, following the updating method of DeepMitosis (<xref ref-type="bibr" rid="B26">Li et al., 2018</xref>). This involves weighting the confidence values <inline-formula id="inf81">
<mml:math id="m85">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="italic">det</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from the detection stage and <inline-formula id="inf82">
<mml:math id="m86">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from the classification stage with a weight parameter <inline-formula id="inf83">
<mml:math id="m87">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to obtain the ultimate target confidence <inline-formula id="inf84">
<mml:math id="m88">
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, as expressed in formula <xref ref-type="disp-formula" rid="e5">5</xref>.<disp-formula id="e5">
<mml:math id="m89">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="italic">det</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf85">
<mml:math id="m90">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the confidence allocation weight, and <xref ref-type="sec" rid="s12">Supplementary Data SB.6</xref> explores the setting of this parameter.</p>
<p>The detailed structural parameters of the classification stage can be found in <xref ref-type="sec" rid="s12">Supplementary Data SA.6.2</xref>.</p>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Experimental setup</title>
<p>All experiments in this study were conducted on a computer running the Ubuntu operating system, utilizing the MMDetection (<xref ref-type="bibr" rid="B9">Chen et al., 2019</xref>) detection framework, complemented by the Pytorch1.9 and the TensorFlow deep learning library. Our computational setup included an Intel(R) Xeon(R) Silver 4110 CPU @ 2.10&#xa0;GHz processor and three GeForce RTX 2080 Ti graphics cards.</p>
<sec id="s3-3-1">
<title>3.3.1 Detection stage</title>
<p>In the MMDetection object detection framework, we utilized ImageNet (<xref ref-type="bibr" rid="B13">Deng et al., 2010</xref>) pre-trained weights to initialize the network&#x2019;s backbone. We employed the data sampling strategy from <xref ref-type="bibr" rid="B38">Piansaddhayanaon et al. (2023)</xref>, randomly selecting 5,000 images of size <inline-formula id="inf86">
<mml:math id="m91">
<mml:mrow>
<mml:mn>512</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>512</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> from each training WSI for training. During training, we set the batch size to 4 and employed random flips, standard photometric augmentations, and other methods to mitigate the risk of overfitting. Stochastic gradient descent (SGD) was the optimizer. The learning rate followed a stepwise constant decay strategy, starting at <inline-formula id="inf87">
<mml:math id="m92">
<mml:mrow>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. After the fifth and seventh epochs, the learning rate was divided by 10, reaching a final decay to <inline-formula id="inf88">
<mml:math id="m93">
<mml:mrow>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The maximum number of training epochs was set to 8.</p>
</sec>
<sec id="s3-3-2">
<title>3.3.2 Classification stage</title>
<p>In the classification stage, initially, we adjust the detected positions of mitotic cells (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>), and detailed experimental settings can be found in <xref ref-type="sec" rid="s12">Supplementary Data SA.5</xref>. Subsequently, we employ EfficientNet-B7 and VGG16 models pre-trained on ImageNet as the backbone of the network. The input image resolution is set to <inline-formula id="inf89">
<mml:math id="m94">
<mml:mrow>
<mml:mn>128</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>128</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, obtained through active learning data sampling methods (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>). We set the batch size to 32 and applied data augmentation techniques, including random translation, random flipping, and standard photometric augmentation. The model is trained using the Adam optimize, with a total of 24,000 training iterations. The initial learning rate is set to <inline-formula id="inf90">
<mml:math id="m95">
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Dynamic updates were implemented by dividing the learning rate by 10&#xa0;at the 15,000th and 21,000th iterations, ultimately decaying to <inline-formula id="inf91">
<mml:math id="m96">
<mml:mrow>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
<sec id="s3-4">
<title>3.4 Model performance evaluation indicators</title>
<p>This paper employs commonly used evaluation metrics, including recall(sensitivity), precision, F1-Score, accuracy, and specificity, as presented in Formulas <xref ref-type="disp-formula" rid="e6">6</xref>&#x2013;<xref ref-type="disp-formula" rid="e10">10</xref>. The F1-Score is obtained by calculating the harmonic mean of precision and recall.<disp-formula id="e6">
<mml:math id="m97">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m98">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m99">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="e9">
<mml:math id="m100">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m101">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>Where TP represents the number of correctly predicted positive samples, FP denotes incorrectly predicted positive samples, FN represents incorrectly predicted negative samples, and TN denotes the number of correctly predicted negative samples.</p>
</sec>
</sec>
<sec sec-type="results" id="s4">
<title>4 Results</title>
<sec id="s4-1">
<title>4.1 Ablation experiments</title>
<sec id="s4-1-1">
<title>4.1.1 Model exploration</title>
<p>As shown in <xref ref-type="table" rid="T1">Table 1</xref>, the performance of the model, combining the detection and classification stages, is superior to that of using only the detection model. In both the integrated model with both detection and classification stages and the case of using only the detection model, the performance of the model is improved with the addition of the DiCoA module.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The ablation experiment comparison between add DiCoA and EfficientNet-B7 &#x2b; VGG16 on the CMC dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Detector</th>
<th rowspan="2" align="center">DiCoA</th>
<th rowspan="2" align="center">EfficientNet-B7 &#x2b; VGG16</th>
<th colspan="3" align="center">Test CMC (%)</th>
</tr>
<tr>
<th align="center">F1</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="center">Cascade R-CNN</td>
<td align="left"/>
<td align="left"/>
<td align="center">68.0</td>
<td align="center">69.8</td>
<td align="center">66.3</td>
</tr>
<tr>
<td align="center">&#x2713;</td>
<td align="left"/>
<td align="center">74.2</td>
<td align="center">72.8</td>
<td align="center">75.8</td>
</tr>
<tr>
<td align="left"/>
<td align="center">&#x2713;<bold>
<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</bold>
</td>
<td align="center">82.2</td>
<td align="center">81.4</td>
<td align="center">83.0</td>
</tr>
<tr>
<td align="center">&#x2713;</td>
<td align="center">&#x2713;<bold>
<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</bold>
</td>
<td align="center">82.9</td>
<td align="center">82.6</td>
<td align="center">83.2</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>Integration of the detection and classification stages.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4-1-2">
<title>4.1.2 Detection stage: comparative analysis of different attention modules</title>
<p>As shown in <xref ref-type="table" rid="T2">Table 2</xref>, on the Cascade R-CNN detection network, our proposed DiCoA attention module, compared to the result without any modifications, exhibited improvements in Recall and F1 by more than 7.5% and 4%, respectively, while experiencing a slight decrease in Precision by 0.1%. In contrast, incorporating the CBAM attention module on the Cascade R-CNN detection network resulted in a 0.5% increase in Recall but led to reductions of 5% and 2% in Precision and F1, respectively. Additionally, for this task, CCNet and SENet attention modules did not yield performance enhancements on the Cascade R-CNN network.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Comparative analysis of various attention modules on cascade R-CNN detection network.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Detector</th>
<th rowspan="2" align="center">Method</th>
<th colspan="3" align="center">Test CMC (%)</th>
</tr>
<tr>
<th align="center">F1</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="5" align="center">Cascade R-CNN</td>
<td align="center">&#x2014;(<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>)<xref ref-type="table-fn" rid="Tfn5">
<sup>a</sup>
</xref>
</td>
<td align="center">70.2</td>
<td align="center">72.9</td>
<td align="center">67.9</td>
</tr>
<tr>
<td align="center">&#x2b; SENet (<xref ref-type="bibr" rid="B17">Hu et al., 2020</xref>)<xref ref-type="table-fn" rid="Tfn3">
<sup>b</sup>
</xref>
</td>
<td align="center">67.6</td>
<td align="center">69.4</td>
<td align="center">66.0</td>
</tr>
<tr>
<td align="center">&#x2b; CCNet (<xref ref-type="bibr" rid="B19">Huang et al., 2018)</xref>
<xref ref-type="table-fn" rid="Tfn3">
<sup>b</sup>
</xref>
</td>
<td align="center">67.7</td>
<td align="center">67.9</td>
<td align="center">67.6</td>
</tr>
<tr>
<td align="center">&#x2b; CBAM (<xref ref-type="bibr" rid="B50">Woo et al., 2018</xref>)<xref ref-type="table-fn" rid="Tfn3">
<sup>b</sup>
</xref>
</td>
<td align="center">68.2</td>
<td align="center">67.9</td>
<td align="center">68.4</td>
</tr>
<tr>
<td align="center">&#x2b; DiCoA<xref ref-type="table-fn" rid="Tfn4">
<sup>c</sup>
</xref>
</td>
<td align="center">74.2</td>
<td align="center">72.8</td>
<td align="center">75.8</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn2">
<label>
<sup>a</sup>
</label>
<p>No improvement methods were implemented.</p>
</fn>
<fn id="Tfn3">
<label>
<sup>b</sup>
</label>
<p>The attention mechanism proposed in the article.</p>
</fn>
<fn id="Tfn4">
<label>
<sup>c</sup>
</label>
<p>The attention module approach proposed by us.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4-1-3">
<title>4.1.3 Classification stage: combining pre-trained models for comparison</title>
<p>After the detection stage, we further investigated the impact of combining different pre-trained models on the performance of the classification stage. According to the results in <xref ref-type="sec" rid="s12">Supplementary Table SB.11</xref>, the combination of two pre-trained models, EfficientNet-B7 and VGG16, achieved optimal performance. Compared to using only the VGG16 model, integrating multiple pre-trained models (InPreMo) resulted in improvements of 2.8%, 0.4%, and 1.6% in Precision, Recall, and F1, respectively. Compared to the EfficientNet-B7 model alone, InPreMo exhibited increases of 1.4% in Recall and 0.3% in F1. Furthermore, when compared to the combination of three pre-trained models (EfficientNet-B7, Resnet50, and VGG16), the combination of two pre-trained models (EfficientNet-B7 and VGG16) suggested higher Recall and F1 by 2.3% and 0.3%, respectively, while experiencing a decrease of 1.8% in Precision. Additionally, <xref ref-type="sec" rid="s12">Supplementary Data SB.9</xref> provides a detailed information on the sensitivity, specificity, and confusion matrix assessments for various classification models.</p>
</sec>
</sec>
<sec id="s4-2">
<title>4.2 Comparison with existing literature</title>
<p>As shown in <xref ref-type="table" rid="T3">Table 3</xref>, our improved method achieved superior results on the CMC dataset compared to existing literature. In contrast to the holistic approach employing the RetinaNet network, our method demonstrates an overall performance improvement of over 5% in Precision, Recall, and F1. Compared to the pipeline of Cascade R-CNN network, our method demonstrates improvements of 0.6%, 1.3%, and 1% in Precision, Recall, and F1, respectively. Additionally, in comparison to the full pipeline of the Faster-RCNN network, our process exhibits an enhancement of 2.4% in Precision, a 0.6% improvement in F1, and a 1.3% decrease in Recall.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Comparison of the proposed method with existing approaches.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Detector</th>
<th rowspan="2" align="center">Method</th>
<th colspan="3" align="center">Test CMC (%)</th>
<th colspan="3" align="center">Test CCMCT (%)</th>
</tr>
<tr>
<th align="center">F1</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="center">RetinaNet (<xref ref-type="bibr" rid="B3">Aubreville et al., 2020</xref>)<xref ref-type="table-fn" rid="Tfn5">
<sup>a</sup>
</xref>
</td>
<td align="center">Detection stage</td>
<td align="center">72.6</td>
<td align="center">69.7</td>
<td align="center">75.8</td>
<td align="center">62.8</td>
<td align="center">57.7</td>
<td align="center">68.8</td>
</tr>
<tr>
<td align="center">Full Pipeline</td>
<td align="center">77.5</td>
<td align="center">77.0</td>
<td align="center">77.9</td>
<td align="center">82.0</td>
<td align="center">82.8</td>
<td align="center">81.2</td>
</tr>
<tr>
<td rowspan="2" align="center">Faster-RCNN (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>)<xref ref-type="table-fn" rid="Tfn5">
<sup>a</sup>
</xref>
</td>
<td align="center">Detection stage</td>
<td align="center">70.4</td>
<td align="center">71.1</td>
<td align="center">69.7</td>
<td align="center">78.2</td>
<td align="center">78.5</td>
<td align="center">77.9</td>
</tr>
<tr>
<td align="center">Full Pipeline</td>
<td align="center">82.3</td>
<td align="center">80.2</td>
<td align="center">84.5</td>
<td align="center">83.2</td>
<td align="center">83.0</td>
<td align="center">83.4</td>
</tr>
<tr>
<td rowspan="2" align="center">Cascade R-CNN (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>)<xref ref-type="table-fn" rid="Tfn5">
<sup>a</sup>
</xref>
</td>
<td align="center">Detection stage</td>
<td align="center">70.2</td>
<td align="center">72.9</td>
<td align="center">67.9</td>
<td align="center">75.8</td>
<td align="center">76.5</td>
<td align="center">75.1</td>
</tr>
<tr>
<td align="center">Full Pipeline</td>
<td align="center">81.9</td>
<td align="center">82.0</td>
<td align="center">81.9</td>
<td align="center">82.9</td>
<td align="center">83.2</td>
<td align="center">82.6</td>
</tr>
<tr>
<td rowspan="2" align="center">Cascade R-CNN</td>
<td align="center">Detection stage<xref ref-type="table-fn" rid="Tfn6">
<sup>b</sup>
</xref>
</td>
<td align="center">74.2</td>
<td align="center">72.8</td>
<td align="center">75.8</td>
<td align="center">77.4</td>
<td align="center">77.6</td>
<td align="center">77.2</td>
</tr>
<tr>
<td align="center">Full Pipeline<xref ref-type="table-fn" rid="Tfn7">
<sup>c</sup>
</xref>
</td>
<td align="center">82.9</td>
<td align="center">82.6</td>
<td align="center">83.2</td>
<td align="center">83.0</td>
<td align="center">83.2</td>
<td align="center">82.9</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn5">
<label>
<sup>a</sup>
</label>
<p>Results obtained from the article.</p>
</fn>
<fn id="Tfn6">
<label>
<sup>b</sup>
</label>
<p>Incorporating our attention module.</p>
</fn>
<fn id="Tfn7">
<label>
<sup>c</sup>
</label>
<p>The final results obtained by our approach.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>After incorporating the improved DiCoA attention module, compared to the Cascade R-CNN detection network, the detection stage exhibited significant improvements of 7.9% and 4% in Recall and F1, respectively. Relative to the Faster R-CNN and the RetinaNet methods, notable enhancements of 3.8% and 1.6% in F1 were observed. Significance testing using a T-test, presented in <xref ref-type="sec" rid="s12">Supplementary Data SB.4</xref>, indicated <italic>p</italic> values &#x3c;0.001 for the F1, demonstrating the statistical significance of using the Cascade R-CNN with the added DiCoA attention module over other methods.</p>
<p>Furthermore, we conducted additional evaluation of the model using the CCMCT dataset. Our approach achieved the best performance in Precision compared to the baseline model. Although the F1 score and Recall were slightly lower by 0.2% and 0.5%, respectively, compared to the performance obtained with the Faster-RCNN model, our method still maintained an advantage over other benchmark models.</p>
<p>The above results indicate that our method enhances the detection performance of mitotic cells.</p>
</sec>
<sec id="s4-3">
<title>4.3 End-to-end evaluation experiment</title>
<p>In an end-to-end setting, following the definition (<xref ref-type="bibr" rid="B33">Meuten et al., 2015</xref>) of mitotic cell counting, we determined the region with the highest predicted mitotic cell count (High-Power Field, HPF) (<xref ref-type="bibr" rid="B5">Bertram et al., 2020</xref>) by counting mitotic shapes in 10 high-power fields (HPFs) of <inline-formula id="inf92">
<mml:math id="m102">
<mml:mrow>
<mml:mn>2.37</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msup>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> each, represented by rectangular windows of size <inline-formula id="inf93">
<mml:math id="m103">
<mml:mrow>
<mml:mn>7110</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>5333</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> pixels. Once the HPF region for the WSI was identified, mitotic cell counting was performed either in a fully automated (GA) manner or through a human-machine interactive approach (GB). Under the fully automated setting, the predicted mitotic cell count in the selected HPF was used as the final mitotic cell count. In the human-machine interactive setting, mitotic cell counting was determined based on the annotated mitotic shapes in the selected HPF. <xref ref-type="table" rid="T4">Table 4</xref> reports the Mean Absolute Percentage Error (MAPE) and Mean Absolute Error (MAE) at the prediction threshold with the lowest MAPE, indicating a significant improvement in mitotic cell counting on the CMC dataset in a human-machine collaborative environment.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>The end-to-end performance of the proposed method, as evaluated on the CMC dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<td rowspan="2" align="center">Dataset</td>
<td rowspan="2" align="center">Method</td>
<td colspan="2" align="center">GA</td>
<td colspan="2" align="center">GB</td>
</tr>
<tr>
<td align="center">MAPE</td>
<td align="center">MAE</td>
<td align="center">MAPE</td>
<td align="center">MAE</td>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="center">CMC</td>
<td align="center">ReCasNet (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>)<xref ref-type="table-fn" rid="Tfn8">
<sup>a</sup>
</xref>
</td>
<td align="center">5.6</td>
<td align="center">1.9</td>
<td align="center">5.6</td>
<td align="center">1.6</td>
</tr>
<tr>
<td align="center">Ours</td>
<td align="center">5.8</td>
<td align="center">2.0</td>
<td align="center">4.3</td>
<td align="center">0.9</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn8">
<label>
<sup>a</sup>
</label>
<p>Results obtained from the article.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>To construct a more accurate model for mitotic cell detection, we devised a two-stage (detection and classification) task model. In the detection stage, we innovatively designed the DiCoA attention module. In the classification stage, we ingeniously proposed a method that integrates multiple pre-trained models to identify mitotic cells. We achieved improved performance on the CMC dataset.</p>
<p>Attention mechanisms are employed to capture crucial features in data, leading to a significant enhancement in model performance (<xref ref-type="bibr" rid="B7">Brauwers and Frasincar, 2022</xref>). Despite the diverse types of attention mechanisms proposed, there is limited literature on how to choose an appropriate attention mechanism for mitotic cell identification in cancer. Therefore, we investigated the application of some commonly used attention mechanisms [SENet, which assists the network in automatically capturing the importance of each feature channel (<xref ref-type="bibr" rid="B17">Hu et al., 2020</xref>); CCNet, which helps the network capture long-range dependencies between feature pixels (<xref ref-type="bibr" rid="B19">Huang et al., 2018</xref>); CBAM, which enhances attention in both spatial and channel dimensions (<xref ref-type="bibr" rid="B50">Woo et al., 2018</xref>)] in the task of mitotic cell recognition. As shown in <xref ref-type="table" rid="T2">Table 2</xref>, while these methods bring varying degrees of performance improvement in their respective domains, their ability to enhance the extraction of advanced features related to mitotic cells is limited. To better extract features of mitotic cells, we proposed a novel DiCoA module to capture remote dependencies between features of mitotic cells. As shown in <xref ref-type="table" rid="T1">Table 1</xref>, the use of the DiCoA attention module benefits the model in both the single detection stage and the combined detection and classification stages. The introduction of DiCoA reduces false negatives and false positives in mitotic predictions (<xref ref-type="sec" rid="s12">Supplementary Data SB.7</xref>). Simultaneously, as demonstrated in <xref ref-type="table" rid="T2">Table 2</xref>, the overall performance (Precision, Recall, and F1) with the inclusion of the DiCoA attention module consistently exceeds 72%, while combining CBAM, SENet, and CCNet attention modules yields an overall performance of only 69%.</p>
<p>In previous studies, researches (<xref ref-type="bibr" rid="B24">Irshad et al., 2013</xref>; <xref ref-type="bibr" rid="B23">2014b</xref>; <xref ref-type="bibr" rid="B29">Lu and Mandal, 2014</xref>; <xref ref-type="bibr" rid="B37">Paul and Mukherjee, 2015</xref>; <xref ref-type="bibr" rid="B35">Nateghi et al., 2017</xref>) extracted features of mitotic cells manually and subsequently employed machine learning methods for mitotic cell identification. While these methods exhibited remarkable interpretability, they necessitated extensive data preprocessing and feature engineering. In contrast, our approach employs an end-to-end algorithm, leveraging the DiCoA attention mechanism and pre-trained models for enhanced feature extraction and application, thereby improving model performance. With the rise of deep learning, it has been applied in mitotic cell recognition (<xref ref-type="bibr" rid="B10">Cire&#x15f;an et al., 2013</xref>; <xref ref-type="bibr" rid="B52">Zerhouni et al., 2017</xref>; <xref ref-type="bibr" rid="B26">Li et al., 2018</xref>; <xref ref-type="bibr" rid="B2">Alom et al., 2020</xref>). Two fully annotated WSI datasets CCMCT and CMC were introduced, and mitotic cell detection was performed using RetinaNet, followed by classification using ResNet18, achieving a baseline performance. To address the challenge of inconsistent data distribution between detection and classification networks, an improved two-stage framework, ReCasNet (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>) was proposed for mitotic detection in CCMCT and CMC datasets. Despite promising results on CCMCT and CMC datasets in existing studies, considering the complexity of mitotic classification detection and model training, the full potential of model performance has yet to be fully explored. As shown in <xref ref-type="sec" rid="s12">Supplementary Data SB.10</xref>, we have summarized and organized various approaches in this field (<xref ref-type="bibr" rid="B10">Cire&#x15f;an et al., 2013</xref>; <xref ref-type="bibr" rid="B1">Albarqouni et al., 2016</xref>; <xref ref-type="bibr" rid="B52">Zerhouni et al., 2017</xref>; <xref ref-type="bibr" rid="B3">Aubreville et al., 2020</xref>; <xref ref-type="bibr" rid="B40">Sebai et al., 2020</xref>; <xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>). With an increase in the number of data samples, the performance of single-stage models is limited, and the adoption of two-stage models can further enhance model performance. However, not all two-stage models yield satisfactory results, indicating the need for further exploration. To enhance the model&#x2019;s performance in mitotic cell recognition and fully exploit the potential of deep learning methods, we developed the DiCoA module, combined with FPN, to identify mitotic cells with diverse scales and shapes. Additionally, we introduced the InPreMo method for fine-grained mitotic classification. As shown in <xref ref-type="table" rid="T3">Table 3</xref>, compared to the best results of existing research on the CMC dataset (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>), our approach achieved an improvement of over 0.5% in Precision and F1. In the detection stage, we introduced the DiCoA module on the Cascade R-CNN network. Compared to the use of Cascade R-CNN and Faster-RCNN networks (<xref ref-type="bibr" rid="B38">Piansaddhayanaon et al., 2023</xref>), our approach demonstrated an improvement of over 6% in Recall and over 3.5% in F1. Finally, we evaluated our method in an end-to-end setting. In a human-machine collaborative scenario, our approach, denoted as MC, exhibited a 43.8% reduction in Mean Absolute Error (MAE) (see <xref ref-type="table" rid="T4">Table 4</xref>).</p>
<p>To enhance the performance in the classification stage, manually extracted features were fused with those obtained from CNN into three classifiers (<xref ref-type="bibr" rid="B47">Wang et al., 2014</xref>), achieving improved performance while minimizing computational resource demands. However, manual feature extraction requires domain-specific expertise and often struggles to adapt to large-scale datasets. A deep belief network with multiple classifiers (<xref ref-type="bibr" rid="B4">Beevi et al., 2017</xref>) was proposed to segment nuclear regions from clinical images. This approach utilizes multiple classifiers and determines the final outcome through majority voting, resulting in enhanced performance. However, precise nuclear segmentation is required for training effective classifiers, and the training of multiple classification models is complex. To address this, we propose a straightforward multi-pre-trained fusion method, combining two distinct pre-trained models, EfficientNet-B7 and VGG16. Compared to using the VGG16 model alone, our approach suggested improvements of over 1% in Precision and F1. Additionally, relative to the EfficientNet-B7 model, we achieved increases of over 0.2% in Recall and F1. These results indicate that the InPreMo method can effortlessly integrate different pre-trained models, leading to effective performance enhancement.</p>
<p>In the detection stage, we compared the results of multiple detection models (<xref ref-type="sec" rid="s12">Supplementary Data SB.1</xref>) and ultimately selected Cascade R-CNN, which demonstrated the best performance, as our detection model. When utilizing the InPreMo approach, the relationship between the number of stacked models and performance is not linear. As shown in <xref ref-type="sec" rid="s12">Supplementary Data SB.8</xref>, compared to the ensemble of models utilizing EfficientNet-B7, Resnet50, and VGG16 pre-trained methods, combining EfficientNet-B7 and VGG16 pre-trained models yielded a more significant performance improvement while reducing model complexity. We also attempted the advanced CNN classification model ConvNeXt (<xref ref-type="bibr" rid="B28">Liu et al., 2022</xref>), but its performance in this task was limited and, therefore, not included. Additionally, constrained by computational resources, we conducted model performance evaluation only on the relatively smaller CMC dataset. Furthermore, despite our method achieving a modest improvement of only 0.5% over the best results from existing research, considering the ubiquity of our approach and the intricate diversity of mitotic cells, our results are deemed acceptable.</p>
<p>It is noteworthy that, when updating the feature map and bounding box confidence of mitotic cells using DiCoA, we found the optimal threshold for mitotic cell bounding boxes to be 0.48 (refer to <xref ref-type="sec" rid="s12">Supplementary Data SB.3</xref>). The reason for this is that, as shown in Eq. <xref ref-type="disp-formula" rid="e4">4</xref>, we add attention scores from the feature map to the bounding box confidence of the original network, giving it a weight of 0.5. This changes the network&#x2019;s confidence.</p>
<p>Updating the confidence of mitotic cell bounding boxes with DiCoA may lead to a decrease in box confidence. If adaptation to other domains is required, it may be worth considering not updating the confidence of the targets. Additionally, the InPreMo method necessitates the selection of an appropriate model depending on the specific task, which warrants further exploration in other studies. Moreover, in terms of model complexity, the introduction of both DiCoA and InPreMo tends to increase the model&#x2019;s complexity to some extent. Although our enhancements have improved the model&#x2019;s capability to extract features related to mitotic cells, further research and optimization are still required to enhance the performance of the network in mitotic cell recognition. Furthermore, in upcoming research, we will further consider addressing variability both between and within observers to ensure the accuracy and reliability of the data.</p>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>We developed the DilCasNet model for more accurate identification of mitotic cells by introducing two key improvements to the two-stage mitotic cell detection method. Firstly, we proposed the DiCoA module with sparse global attention, effectively enhancing the detection network&#x2019;s ability to capture long-range dependencies between features of mitotic cells. This enables the model to better recognize mitotic cells of varying sizes and shapes, reducing false-negative and false-positive predictions while significantly improving overall performance. Secondly, we ingeniously integrated the EfficientNet-B7 and VGG16 pre-trained models, enhancing the model&#x2019;s performance in the classification stage. This approach provides a novel choice for current classification networks. Our method demonstrated improved performance in detecting mitotic cells on the CMC dataset.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s12">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s8">
<title>Author contributions</title>
<p>ZL: Methodology, Writing&#x2013;original draft. XKL: Data curation, Methodology, Writing&#x2013;original draft, Writing&#x2013;review and editing. WW: Data curation, Formal Analysis, Writing&#x2013;review and editing. HL: Data curation, Writing&#x2013;review and editing. XT: Formal Analysis, Writing&#x2013;review and editing. CZ: Writing&#x2013;review and editing. FX: Writing&#x2013;review and editing. BL: Writing&#x2013;review and editing. YJ: Conceptualization, Funding acquisition, Writing&#x2013;review and editing. XWL: Writing&#x2013;review and editing. WX: Conceptualization, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was partially supported by National Nature Science Foundation (62073270), State Ethnic Affairs Commission Innovation Research Team, and Innovative Research Team of the Education Department of Sichuan Province (15TD0050). This research was supported by the Fundamental Research Funds for Central University, Southwest Minzu University (2022NYXXS111).</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>Author BL was employed by Sichuan Huhui Software Co., LTD.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fphys.2024.1337554/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fphys.2024.1337554/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Albarqouni</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Baur</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Achilles</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Belagiannis</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Demirci</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Navab</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>AggNet: deep learning from crowds for mitosis detection in breast cancer histology images</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>35</volume>, <fpage>1313</fpage>&#x2013;<lpage>1321</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2016.2528120</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alom</surname>
<given-names>M. Z.</given-names>
</name>
<name>
<surname>Aspiras</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Taha</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Bowen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Asari</surname>
<given-names>V. K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>MitosisNet: end-to-end mitotic cell detection by multi-task learning</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>68695</fpage>&#x2013;<lpage>68710</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.2983995</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aubreville</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bertram</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Donovan</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Marzahl</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Maier</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Klopfleisch</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A completely annotated whole slide image dataset of canine breast cancer to aid human breast cancer research</article-title>. <source>Sci. Data</source> <volume>71</volume> (<issue>7</issue>), <fpage>417</fpage>&#x2013;<lpage>510</lpage>. <pub-id pub-id-type="doi">10.1038/s41597-020-00756-z</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beevi</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Nair</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Bindu</surname>
<given-names>G. R.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A multi-classifier system for automatic mitosis detection in breast histopathology images using deep belief networks</article-title>. <source>IEEE J. Transl. Eng. Heal. Med.</source> <volume>5</volume>, <fpage>4300211</fpage>. <pub-id pub-id-type="doi">10.1109/JTEHM.2017.2694004</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bertram</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Aubreville</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gurtner</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bartel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Corner</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Dettwiler</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Computerized calculation of mitotic count distribution in canine cutaneous mast cell tumor sections: mitotic count is area dependent</article-title>. <source>Vet. Pathol.</source> <volume>57</volume>, <fpage>214</fpage>&#x2013;<lpage>226</lpage>. <pub-id pub-id-type="doi">10.1177/0300985819890686</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bertram</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Aubreville</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Marzahl</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Maier</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Klopfleisch</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A large-scale dataset for mitotic figure assessment on whole slide images of canine cutaneous mast cell tumor</article-title>. <source>Sci. Data</source> <volume>61</volume> (<issue>6</issue>), <fpage>274</fpage>&#x2013;<lpage>279</lpage>. <pub-id pub-id-type="doi">10.1038/s41597-019-0290-4</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brauwers</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Frasincar</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A general survey on attention mechanisms in deep learning</article-title>. <source>IEEE Trans. Knowl. Data Eng.</source> <volume>35</volume>, <fpage>3279</fpage>&#x2013;<lpage>3298</lpage>. <pub-id pub-id-type="doi">10.1109/TKDE.2021.3126456</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Vasconcelos</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Cascade R-CNN: delving into high quality object detection</article-title>. <source>Proc. IEEE Comput. Soc. Conf. Comput. Vis. Pattern Recognit.</source>, <fpage>6154</fpage>&#x2013;<lpage>6162</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2018.00644</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>MMDetection: open MMLab detection toolbox and benchmark</article-title>. <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1906.07155">https://arxiv.org/abs/1906.07155</ext-link>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cire&#x15f;an</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Giusti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gambardella</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Schmidhuber</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Mitosis detection in breast cancer histology images with deep neural networks</article-title>. <source>Med. Image Comput. Comput. Assist. Interv.</source> <volume>16</volume>, <fpage>411</fpage>&#x2013;<lpage>418</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-642-40763-5_51</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cree</surname>
<given-names>I. A.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>P. H.</given-names>
</name>
<name>
<surname>Travis</surname>
<given-names>W. D.</given-names>
</name>
<name>
<surname>Wesseling</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Yagi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>V. A.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Counting mitoses: SI(ze) matters</article-title>. <source>Mod. Pathol.</source> <volume>349</volume> (<issue>34</issue>), <fpage>1651</fpage>&#x2013;<lpage>1657</lpage>. <pub-id pub-id-type="doi">10.1038/s41379-021-00825-7</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>L. J.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y. Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y. W.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Molecular features and clinical implications of the heterogeneity in Chinese patients with HER2-low breast cancer</article-title>. <source>Nat. Commun.</source> <volume>141</volume> (<issue>14</issue>), <fpage>5112</fpage>&#x2013;<lpage>5113</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-023-40715-x</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Deng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Socher</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.-J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Fei-Fei</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>ImageNet: a large-scale hierarchical image database</article-title>,&#x201d; in <conf-name>2009 IEEE Conference on Computer Vision and Pattern Recognition</conf-name>, <conf-loc>Miami, FL, USA</conf-loc>, <conf-date>June, 2010</conf-date>, <fpage>248</fpage>&#x2013;<lpage>255</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2009.5206848</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gurcan</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Boucheron</surname>
<given-names>L. E.</given-names>
</name>
<name>
<surname>Can</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Madabhushi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rajpoot</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Yener</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Histopathological image analysis: a review</article-title>. <source>IEEE Rev. Biomed. Eng.</source> <volume>2</volume>, <fpage>147</fpage>&#x2013;<lpage>171</lpage>. <pub-id pub-id-type="doi">10.1109/RBME.2009.2034865</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Hassani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Dilated neighborhood attention transformer</article-title>. <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2209.15001">https://arxiv.org/abs/2209.15001</ext-link>.</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Deep residual learning for image recognition</article-title>,&#x201d; in <conf-name>Proc. IEEE Comput. Soc. Conf. Comput. Vis. Pattern Recognit</conf-name>, <conf-loc>Las Vegas, NV, USA</conf-loc>, <conf-date>June, 2016</conf-date>, <fpage>770</fpage>&#x2013;<lpage>778</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Albanie</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Squeeze-and-Excitation networks</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>42</volume>, <fpage>2011</fpage>&#x2013;<lpage>2023</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2019.2913372</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Van Der Maaten</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Weinberger</surname>
<given-names>K. Q.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Densely connected convolutional networks</article-title>,&#x201d; in <conf-name>Proc. - 30th IEEE Conf. Comput. Vis. Pattern Recognition, CVPR 2017</conf-name>, <conf-loc>Honolulu, Hawaii</conf-loc>, <conf-date>January, 2017</conf-date>, <fpage>2261</fpage>&#x2013;<lpage>2269</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2017.243</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>CCNet: criss-cross attention for semantic segmentation</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>45</volume>, <fpage>6896</fpage>&#x2013;<lpage>6908</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2020.3007032</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ibrahim</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lashen</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Toss</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mihai</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rakha</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Assessment of mitotic activity in breast cancer: revisited in the digital pathology era</article-title>. <source>J. Clin. Pathol.</source> <volume>75</volume>, <fpage>365</fpage>&#x2013;<lpage>372</lpage>. <pub-id pub-id-type="doi">10.1136/JCLINPATH-2021-207742</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ibrahim</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lashen</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Katayama</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mihai</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ball</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Toss</surname>
<given-names>M. S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Defining the area of mitoses counting in invasive breast cancer using whole slide image</article-title>. <source>Mod. Pathol.</source> <volume>356</volume> (<issue>35</issue>), <fpage>739</fpage>&#x2013;<lpage>748</lpage>. <pub-id pub-id-type="doi">10.1038/s41379-021-00981-w</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Irshad</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gouaillard</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Roux</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Racoceanu</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2014a</year>). <article-title>Multispectral band selection and spatial characterization: application to mitosis detection in breast cancer histopathology</article-title>. <source>Comput. Med. Imaging Graph.</source> <volume>38</volume>, <fpage>390</fpage>&#x2013;<lpage>402</lpage>. <pub-id pub-id-type="doi">10.1016/J.COMPMEDIMAG.2014.04.003</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Irshad</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gouaillard</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Roux</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Racoceanu</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2014b</year>). &#x201c;<article-title>Spectral band selection for mitosis detection in histopathology</article-title>,&#x201d; in <conf-name>2014 IEEE 11th Int. Symp. Biomed. Imaging, ISBI</conf-name>, <conf-loc>Beijing, China</conf-loc>, <conf-date>April, 2014</conf-date>, <fpage>1279</fpage>&#x2013;<lpage>1282</lpage>. <pub-id pub-id-type="doi">10.1109/ISBI.2014.6868110</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Irshad</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jalali</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Roux</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Racoceanu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hwee</surname>
<given-names>L. J.</given-names>
</name>
<name>
<surname>Naour</surname>
<given-names>G.Le</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Automated mitosis detection using texture, SIFT features and HMAX biologically inspired approach</article-title>. <source>J. Pathol. Inf.</source> <volume>4</volume>, <fpage>12</fpage>. <pub-id pub-id-type="doi">10.4103/2153-3539.109870</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lecun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>Nat</source> <volume>521</volume>, <fpage>436</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Latecki</surname>
<given-names>L. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>DeepMitosis: mitosis detection via deep detection, verification and segmentation networks</article-title>. <source>Med. Image Anal.</source> <volume>45</volume>, <fpage>121</fpage>&#x2013;<lpage>133</lpage>. <pub-id pub-id-type="doi">10.1016/J.MEDIA.2017.12.002</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>T.-Y.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hariharan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Belongie</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Feature Pyramid networks for object detection</article-title>,&#x201d; in <conf-name>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Honolulu, HI, USA</conf-loc>, <conf-date>July, 2016</conf-date>. <pub-id pub-id-type="doi">10.48550/arxiv.1612.03144</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Feichtenhofer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Darrell</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>A ConvNet for the 2020s</article-title>,&#x201d; in <conf-name>Proc. IEEE Comput. Soc. Conf. Comput. Vis. Pattern Recognit</conf-name>, <conf-loc>New Orleans, LA, USA</conf-loc>, <conf-date>June, 2022</conf-date>, <fpage>11966</fpage>&#x2013;<lpage>11976</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01167</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mandal</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Toward automatic mitotic cell detection and segmentation in multispectral histopathological images</article-title>. <source>IEEE J. Biomed. Heal. Inf.</source> <volume>18</volume>, <fpage>594</fpage>&#x2013;<lpage>605</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2013.2277837</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ludovic</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Daniel</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Nicolas</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Maria</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Humayun</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Jacques</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Mitosis detection in breast cancer histological images an ICPR 2012 contest</article-title>. <source>J. Pathol. Inf.</source> <volume>4</volume>, <fpage>8</fpage>. <pub-id pub-id-type="doi">10.4103/2153-3539.112693</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahmood</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Arsalan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Owais</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>K. R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Artificial intelligence-based mitosis detection in breast cancer histopathology images using faster R-CNN and deep CNNs</article-title>. <source>J. Clin. Med.</source> <volume>9</volume>, <fpage>749</fpage>. <pub-id pub-id-type="doi">10.3390/JCM9030749</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mathew</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kini</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Rajan</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Computational methods for automated mitosis detection in histopathology images: a review</article-title>. <source>Biocybern. Biomed. Eng.</source> <volume>41</volume>, <fpage>64</fpage>&#x2013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.1016/J.BBE.2020.11.005</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meuten</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Moore</surname>
<given-names>F. M.</given-names>
</name>
<name>
<surname>George</surname>
<given-names>J. W.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Mitotic count and the field of view area: time to standardize</article-title>. <source>Vet. Pathol.</source> <volume>53</volume>, <fpage>7</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1177/0300985815593349</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="web">
<collab>MITOS-ATYPIA-14</collab> (<year>2014</year>). <article-title>Mitos-Atypia-14-Dataset</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://mitos-atypia-14.grand-challenge.org/Dataset/">https://mitos-atypia-14.grand-challenge.org/Dataset/</ext-link>.</comment>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nateghi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Danyali</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Helfroush</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Maximized inter-class weighted mean for fast and accurate mitosis cells detection in breast cancer histopathology images</article-title>. <source>J. Med. Syst.</source> <volume>41</volume>, <fpage>146</fpage>&#x2013;<lpage>215</lpage>. <pub-id pub-id-type="doi">10.1007/s10916-017-0773-9</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>A survey on transfer learning</article-title>. <source>IEEE Trans. Knowl. Data Eng.</source> <volume>22</volume>, <fpage>1345</fpage>&#x2013;<lpage>1359</lpage>. <pub-id pub-id-type="doi">10.1109/TKDE.2009.191</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paul</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mukherjee</surname>
<given-names>D. P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Mitosis detection for invasive breast cancer grading in histopathological images</article-title>. <source>IEEE Trans. Image Process.</source> <volume>24</volume>, <fpage>4041</fpage>&#x2013;<lpage>4054</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2015.2460455</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Piansaddhayanaon</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Santisukwongchote</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shuangshoti</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Sriswasdi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chuangsuwanich</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>ReCasNet: improving consistency within the two-stage mitosis detection framework</article-title>. <source>Artif. Intell. Med.</source> <volume>135</volume>, <fpage>102462</fpage>. <pub-id pub-id-type="doi">10.1016/J.ARTMED.2022.102462</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Faster R-CNN: towards real-time object detection with region proposal networks</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>39</volume>, <fpage>1137</fpage>&#x2013;<lpage>1149</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2016.2577031</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sebai</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Al-Fadhli</surname>
<given-names>S. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>PartMitosis: a partially supervised deep learning framework for mitosis detection in breast cancer histopathology images</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>45133</fpage>&#x2013;<lpage>45147</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.2978754</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Simonyan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zisserman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Very deep convolutional networks for large-scale image recognition</article-title>,&#x201d; in <conf-name>3rd Int. Conf. Learn. Represent. ICLR 2015 - Conf. Track Proc.</conf-name>, <conf-loc>San Diego, CA, USA</conf-loc>, <conf-date>May, 2015</conf-date>.</citation>
</ref>
<ref id="B42">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>Q. V.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>EfficientNet: rethinking model scaling for convolutional neural networks</article-title>,&#x201d; in <conf-name>36th International Conference on Machine Learning</conf-name>, <conf-loc>Long Beach, CA, USA</conf-loc>, <conf-date>June, 2019</conf-date>, <fpage>6105</fpage>&#x2013;<lpage>6114</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tran</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Kondrashova</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Bradley</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Williams</surname>
<given-names>E. D.</given-names>
</name>
<name>
<surname>Pearson</surname>
<given-names>J. V.</given-names>
</name>
<name>
<surname>Waddell</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep learning in cancer diagnosis, prognosis and treatment selection</article-title>. <source>Genome Med.</source> <volume>13</volume>, <fpage>152</fpage>. <pub-id pub-id-type="doi">10.1186/S13073-021-00968-X</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Udousoro</surname>
<given-names>I. C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Machine learning: a review</article-title>. <source>Semicond. Sci. Inf. Devices</source> <volume>2</volume>, <fpage>5</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.30564/SSID.V2I2.1931</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veta</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Heng</surname>
<given-names>Y. J.</given-names>
</name>
<name>
<surname>Stathonikos</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bejnordi</surname>
<given-names>B. E.</given-names>
</name>
<name>
<surname>Beca</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wollmann</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Predicting breast tumor proliferation from whole-slide images: the TUPAC16 challenge</article-title>. <source>Med. Image Anal.</source> <volume>54</volume>, <fpage>111</fpage>&#x2013;<lpage>121</lpage>. <pub-id pub-id-type="doi">10.1016/J.MEDIA.2019.02.012</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veta</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>van Diest</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Willems</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Madabhushi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cruz-Roa</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Assessment of algorithms for mitosis detection in breast cancer histopathology images</article-title>. <source>Med. Image Anal.</source> <volume>20</volume>, <fpage>237</fpage>&#x2013;<lpage>248</lpage>. <pub-id pub-id-type="doi">10.1016/J.MEDIA.2014.11.010</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cruz-Roa</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Basavanhally</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gilmore</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shih</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Feldman</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Cascaded ensemble of convolutional neural networks and handcrafted features for mitosis detection</article-title>. <source>Med. Imaging 2014 Digit. Pathol.</source> <volume>9041</volume>, <fpage>90410B</fpage>. <pub-id pub-id-type="doi">10.1117/12.2043902</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>3D soma detection in large-scale whole brain images via a two-stage neural network</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>42</volume>, <fpage>148</fpage>&#x2013;<lpage>157</lpage>. <pub-id pub-id-type="doi">10.1109/tmi.2022.3206605</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>INet: convolutional networks for biomedical image segmentation</article-title>. <source>IEEE Access</source> <volume>9</volume>, <fpage>16591</fpage>&#x2013;<lpage>16603</lpage>. <pub-id pub-id-type="doi">10.1109/access.2021.3053408</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Woo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J. Y.</given-names>
</name>
<name>
<surname>Kweon</surname>
<given-names>I. S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>CBAM: convolutional Block attention module</article-title>,&#x201d; in <conf-name>Proc. Eur. Conf. Comput. Vis. 11211 LNCS</conf-name>, <conf-loc>Munich, Germany</conf-loc>, <conf-date>September, 2018</conf-date>, <fpage>3</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-01234-2_1</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Global trends and forecasts of breast cancer incidence and deaths</article-title>. <source>Sci. Data</source> <volume>10</volume>, <fpage>334</fpage>&#x2013;<lpage>410</lpage>. <pub-id pub-id-type="doi">10.1038/s41597-023-02253-5</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zerhouni</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lanyi</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Viana</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gabrani</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Wide residual networks for mitosis detection</article-title>. <source>Proc. - Int. Symp. Biomed. Imaging</source>, <fpage>924</fpage>&#x2013;<lpage>928</lpage>. <pub-id pub-id-type="doi">10.1109/ISBI.2017.7950667</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>