<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1748468</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Orchestrating segment anything models to accelerate segmentation annotation on agricultural image datasets</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Oehme</surname>
<given-names>Leon H.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3283993"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Boysen</surname>
<given-names>Jonas</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Zhangkai</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Stein</surname>
<given-names>Anthony</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>M&#x00FC;ller</surname>
<given-names>Joachim</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1518644"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>Institute of Agricultural Engineering, Tropics and Subtropics Group, University of Hohenheim</institution>, <city>Stuttgart</city>, <country country="de">Germany</country></aff>
<aff id="aff2"><label>2</label><institution>Institute of Agricultural Engineering, Department of Artificial Intelligence in Agricultural Engineering, University of Hohenheim</institution>, <city>Stuttgart</city>, <country country="de">Germany</country></aff>
<author-notes>
<corresp id="c001"><label>&#x002A;</label>Correspondence: Leon H. Oehme, <email xlink:href="mailto:leon.oehme@uni-hohenheim.de">leon.oehme@uni-hohenheim.de</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-01-22">
<day>22</day>
<month>01</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1748468</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>11</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>27</day>
<month>12</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>29</day>
<month>12</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2026 Oehme, Boysen, Wu, Stein and M&#x00FC;ller.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>Oehme, Boysen, Wu, Stein and M&#x00FC;ller</copyright-holder>
<license>
<ali:license_ref start_date="2026-01-22">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>Increasingly many applications of machine vision and artificial intelligence (AI) can be observed in agriculture. Yet, high-quality training data remains a bottleneck in the development of many AI solutions, particularly for image segmentation. Therefore, ARAMSAM (agricultural rapid annotation module based on segment anything models) was developed, a user interface that orchestrates the pre-labelling capabilities of both the segment anything models (SAM 1, SAM 2) and conventional annotation tools. One <italic>in silico</italic> experiment on zero-shot performance of SAM 1 and SAM 2 on three unseen agricultural datasets and another experiment on hyperparameter optimization of the automatic mask generators (AMG) were conducted. In a user experiment, 14 agricultural experts applied ARAMSAM to quantify the reduction of annotation times. SAM 2 benefited greatly from hyperparameter optimization of its AMG. Based on ground-truth masks matched with predicted masks, the <italic>F<sub>2</sub></italic>-score of SAM 2 improved from 0.05 to 0.74, while that of SAM 1 was improved from 0.87 to 0.93. The user interaction time could be reduced to 2.1 s/mask on single images (SAM 1) and to 1.6 s/mask on image sequences (SAM 2) compared to polygon drawing (9.7 s/mask). This study demonstrates the potential of segment anything models as incorporated into ARAMSAM to significantly accelerate the process of segmentation mask annotation in agriculture and other fields. ARAMSAM will be released as open-source software (AGPL-3.0 license) at <uri xlink:href="https://github.com/DerOehmer/ARAMSAM">https://github.com/DerOehmer/ARAMSAM</uri>.</p>
</abstract>
<kwd-group>
<kwd>agriculture</kwd>
<kwd>annotation</kwd>
<kwd>deep learning</kwd>
<kwd>phenotyping</kwd>
<kwd>segment anything model 2</kwd>
<kwd>segmentation</kwd>
<kwd>UAV</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declared that financial support was received for this work and/or its publication. Funded by the Deutsche Forschungsgemeinschaft (DFG, German Research Foundation) &#x2013; 328017493/GRK 2366 (Sino-German International Research Training Group AMAIZE-P). Additionally, a minor part of the project is supported by funds of the Federal Ministry of Food and Agriculture (BMEL) based on a decision of the Parliament of the Federal Republic of Germany. The Federal Office for Agriculture and Food (BLE) provided coordinating support for artificial intelligence (AI) in agriculture as funding organization, grant number 28DK109A20.</funding-statement>
</funding-group>
<counts>
<fig-count count="7"/>
<table-count count="2"/>
<equation-count count="7"/>
<ref-count count="50"/>
<page-count count="14"/>
<word-count count="11287"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>AI in Food, Agriculture and Water</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>In recent years, the rapid development of machine vision based on artificial intelligence (AI) has gained increasing attention in agriculture (<xref ref-type="bibr" rid="ref1">Abbasi et al., 2022</xref>; <xref ref-type="bibr" rid="ref18">Maraveas, 2024</xref>). This becomes especially apparent in the field of plant phenotyping, where AI enables more precise and efficient analysis of plant traits (<xref ref-type="bibr" rid="ref12">Farooq et al., 2024</xref>; <xref ref-type="bibr" rid="ref36">Sheikh et al., 2024</xref>; <xref ref-type="bibr" rid="ref42">Visakh et al., 2024</xref>). However, the application of AI often necessitates large quantities of labeled data, the preparation of which demands substantial time and effort (<xref ref-type="bibr" rid="ref27">Paton et al., 2024</xref>). Creating accurate labels in agriculture often requires specialized knowledge, such as determining whether a pixel belongs to a specific weed type, further increasing the cost of the annotation process. Among annotation tasks, creating segmentation masks is particularly labor-intensive compared to deep learning tasks like classification or object detection.</p>
<p>As a subfield of image segmentation, every object instance of each class is assigned to one mask in instance segmentation. Such instances could be, e.g., single blood cells in a histological exam (<xref ref-type="bibr" rid="ref25">Pal et al., 2024</xref>) or single maize kernels in maize ear phenotyping (<xref ref-type="bibr" rid="ref24">Oury et al., 2022</xref>). Further applications of instance segmentation in plant phenotyping are the segmentation of the grapevine inflorescence (<xref ref-type="bibr" rid="ref21">Moreira et al., 2025</xref>), or the counting of wheat ears (<xref ref-type="bibr" rid="ref8">Dandrifosse et al., 2022</xref>). All these studies have in common that the training and testing of the proposed deep learning models rely heavily on high-quality ground-truth data.</p>
<p>Traditionally, annotation of segmentation masks involved pixel-wise labeling or drawing polygons to create precise masks (<xref ref-type="bibr" rid="ref7">Castrej&#x00F3;n et al., 2017</xref>). More recently, the adoption of AI-driven pre-labeling tools has emerged as a promising approach to accelerate the annotation process. Pre-labeling shifts the role of human annotators from manual labeling to refining AI-generated labels, reducing the effort required for data annotation (<xref ref-type="bibr" rid="ref35">Shao et al., 2024</xref>). A suitable source for pre-labels in segmentation is the recently released foundation models segment anything model 1 (SAM 1) (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>) and its successor, the segment anything model 2 (SAM 2) (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>). Both models were trained and successfully tested on various domains (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>; <xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>). While SAM 1 only predicts masks on individual images (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>), SAM 2 was designed to predict and track masks along video frames (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>). Both models feature an automatic mask generator (AMG), proposing masks without required input, and the prediction of masks based on input prompts such as bounding boxes or points (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>; <xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>). Instead of using SAM 1 for pre-labeling, its prompting capabilities were often applied directly on different phenotyping tasks, such as the segmentation of potato leaves (<xref ref-type="bibr" rid="ref46">Williams et al., 2024</xref>) or for phenotypical measurements on pumpkin, radish, and cucumber (<xref ref-type="bibr" rid="ref48">Zhang et al., 2024</xref>).</p>
<p>In agriculture, images are typically collected from mobile platforms such as unmanned aerial vehicles (UAV) (<xref ref-type="bibr" rid="ref23">Oehme et al., 2022</xref>; <xref ref-type="bibr" rid="ref30">Rejeb et al., 2022</xref>), tractors (<xref ref-type="bibr" rid="ref2">Boysen et al., 2023</xref>) or stationary plant phenotyping systems (<xref ref-type="bibr" rid="ref9">Daviet et al., 2022</xref>; <xref ref-type="bibr" rid="ref15">Kirchgessner et al., 2024</xref>). Here, one or more cameras move relative to one or more objects of interest, resulting in image sequences having varying overlap between images. In scenarios where such overlapping images need to be annotated, a human may need to annotate the same object on multiple images. Photogrammetry allows the orientation and merging of overlapping images, which is often applied in UAV imagery, resulting in orthomosaics (<xref ref-type="bibr" rid="ref30">Rejeb et al., 2022</xref>). Annotators could, e.g., annotate masks on one combined orthomosaic instead of multiple original images. Yet orthomosaics can contain artifacts or distortions (<xref ref-type="bibr" rid="ref17">Manzini et al., 2024</xref>), leading to bad annotations that might affect machine vision applications. In contrast, SAM 2&#x2019;s mask propagation capabilities allow transferring masks from one consecutive image to the next without relying on photogrammetry. SAM 2&#x2019;s design for video segmentation indicates robustness even on complex scenes, whereas photogrammetry assumes scenes do not move between captured images.</p>
<p>Although open-source annotation software, such as LabelMe, has already integrated SAM 1 as a pre-labeling tool (<xref ref-type="bibr" rid="ref43">Wada, 2025</xref>), the effect of such tools on annotation time efficiency has not been studied on agricultural datasets. Similarly, to this date, no systematic optimization of AMG parameter selection has been conducted.</p>
<p>This study investigates the feasibility of using SAM 1 and SAM 2 as a pre-labeling tool to reduce instance segmentation annotation efforts on agricultural datasets. The study serves as a pathway to designing efficient annotation strategies, ranging from encoder selection to AMG hyperparameter optimization to the selection of suitable annotation tools. Therefore, the agricultural rapid annotation module based on segment anything models (ARAMSAM) is proposed, an open-source application built on top of SAM 1 and SAM 2. In this study, three key objectives are addressed:</p>
<list list-type="roman-lower">
<list-item><p>Evaluating the zero-shot performance of SAM 1 and SAM 2 encoders on previously unseen agricultural datasets;</p></list-item>
<list-item><p>Optimizing AMG hyperparameters via a systematic grid search and analyzing its impact on annotation efforts;</p></list-item>
<list-item><p>Quantifying the reduction in user interaction time of SAM-based methods as orchestrated by ARAMSAM and comparing them to polygon drawing as the previous standard method.</p></list-item>
</list>
</sec>
<sec sec-type="materials|methods" id="sec2">
<label>2</label>
<title>Materials and methods</title>
<sec id="sec3">
<label>2.1</label>
<title>Datasets</title>
<p>Three datasets of RGB images, representing a range of common agricultural applications, were included in the experiments: (a) a maize ear dataset (MED), (b) a maize field UAV dataset (MUD) and (c) a soil surface dataset (SOD) (<xref ref-type="fig" rid="fig1">Figure 1</xref>).</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Image datasets: <bold>(a)</bold> Maize ear dataset (MED), <bold>(b)</bold> maize UAV dataset (MUD), <bold>(c)</bold> soil dataset (SOD); blue boxes highlight one segmentation instance as example, with the mask shown in pink.</p>
</caption>
<graphic xlink:href="frai-08-1748468-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Corn kernel close-up showing its texture, young maize plants in neat rows with a highlighted seedling, and soil with highlighted small object. Each section is connected with dashed lines and boxes.</alt-text>
</graphic>
</fig>
<p>Images of the MED, as shown in <xref ref-type="fig" rid="fig1">Figure 1a</xref>, were captured under controlled lighting conditions using an Alvium 1800 C-2050 camera (Allied Vision Technologies GmbH, Stadtroda, Germany) with a resolution of up to 5,376 &#x00D7; 3,672 pixels. The sensor was attached to a Kowa LM8FC24M lens (Kowa Company, Ltd., Nagoya, Japan) with an 8.5&#x202F;mm focal length. Each ear was captured at 50 evenly distributed horizontal positions around the ear, rotating it stepwise by an angle <italic>&#x03C9;</italic> of 7.1&#x00B0; between each image using a motorized rotating platform. To exclude the background area and to limit the annotation time per maize ear, the images were cropped to include only the upper half of the ear. Individual maize kernels represent the target object instances during later experiments.</p>
<p>The MUD comprises images of juvenile maize plants cultivated in two-row plots during a field trial (<xref ref-type="fig" rid="fig1">Figure 1b</xref>). These images were acquired in June 2024 using an UAV DJI M350 (SZ DJI Technology Co., Ltd., Shenzhen, China) operating at an altitude of 20&#x202F;m above ground at an experimental farm of the University of Hohenheim in Stuttgart, Germany. The UAV was equipped with a DJI&#x2019;s Zenmuse P1 sensor (8,192&#x202F;&#x00D7;&#x202F;5,460 pixels) and a P1 50&#x202F;mm lens resulting in a ground sample distance of 3.1&#x202F;mm/pixel. When conducting field experiments, the phenotypic data are usually collected per plot. To simulate the common postprocessing of experimental field data, the images were cropped to show one plot per image. The target instances are individual maize plants, and occluded parts are also included in the segmentation masks. Neither the MED nor the MUD has been published previously.</p>
<p>The SOD, as shown in <xref ref-type="fig" rid="fig1">Figure 1c</xref>, was collected after sowing with a power harrow sowing combination on ploughed fields around Stuttgart, Germany, in October and November 2022. The camera was mounted on the back of the machine and captured the images from a bird&#x2019;s eye view. The images were captured with the SceneScan Pro-system of Nerian vision technologies (Allied Vision Technologies GmbH, Stadtroda, Germany) and were cropped to a size of 512&#x202F;&#x00D7;&#x202F;512 pixels. The ground sampling distance of the images is 1&#x202F;mm/pixel. The dataset has been previously used to model the soil-machine interaction during secondary tillage by utilizing a deep learning model in <xref ref-type="bibr" rid="ref2">Boysen et al. (2023)</xref>. Individual soil clods represent the target instances for segmentation. Neither of the three datasets has been included in the training datasets of SAM 1 or SAM 2.</p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Encoder experiment</title>
<p>The architecture of SAM 1 (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>) and its successor SAM 2 (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>) heavily rely on their image encoders for feature extraction. The encoder constitutes the largest part of the models and has a large influence on the resulting inference speed and segmentation quality. While SAM 1 employs the original vision transformers (ViT) by <xref ref-type="bibr" rid="ref11">Dosovitskiy et al. (2020)</xref> as encoders, SAM 2 is based on less computationally complex hierarchical vision transformers (Hiera) (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>). For encoder selection, the performance of the three released encoders of SAM 1 (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>) (ViT-B, ViT-L, ViT-H) was evaluated. Additionally, the four encoders of SAM 2 (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>) (Hiera-T, Hiera-S, Hiera-B+, Hiera-L) were evaluated in both their initially released version (SAM 2.0) and their updated version (SAM 2.1). To assess segmentation quality, the models were applied to all three datasets (MED, MUD, SOD). Therefore, 10 images and 10 object instances per image were randomly selected and annotated with the polygon feature of ARAMSAM (see chapter 2.4). The geometric median of the respective ground-truth masks, as defined by <xref ref-type="bibr" rid="ref41">Vardi and Zhang (2000)</xref>, was used as a positive point prompt for the model. A positive point prompt indicates to the model where to find a mask at the specific point in the image. Multiple points may be prompted to SAM to generate a mask. In contrast, negative points can be prompted to confine masks or exclude regions from a mask (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>). To quantify segmentation accuracy while accommodating class imbalance between relatively small object instances and the background, the generalized dice score (<italic>GDS</italic>) (<xref ref-type="bibr" rid="ref38">Sudre et al., 2017</xref>) implemented in Monai (1.4) (<xref ref-type="bibr" rid="ref6">Cardoso et al., 2022</xref>) was used as a metric to evaluate the models&#x2019; performance. For this specific two-class case, the <italic>GDS</italic> can be defined for <italic>N</italic> pixels as:</p>
<disp-formula id="E1"><mml:math id="M1"><mml:mi mathvariant="italic">GDS</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:msubsup><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mn>2</mml:mn></mml:msubsup><mml:msub><mml:mi>w</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:msubsup><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mspace width="0.1em"/><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msubsup><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mn>2</mml:mn></mml:msubsup><mml:msub><mml:mi>w</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:msubsup><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:mo stretchy="true">(</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:math><label>(1)</label></disp-formula>
<p>Where <inline-formula><mml:math id="M2"><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mo stretchy="true">{</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="true">}</mml:mo></mml:math></inline-formula> are the ground-truth labels and <inline-formula><mml:math id="M3"><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mo stretchy="true">{</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="true">}</mml:mo></mml:math></inline-formula> are the predicted labels for the class <inline-formula><mml:math id="M4"><mml:mi>l</mml:mi></mml:math></inline-formula> at pixel position <inline-formula><mml:math id="M5"><mml:mi>i</mml:mi></mml:math></inline-formula>. The weight per class <inline-formula><mml:math id="M6"><mml:msub><mml:mi>w</mml:mi><mml:mi>l</mml:mi></mml:msub></mml:math></inline-formula> is defined as:</p>
<disp-formula id="E2"><mml:math id="M7"><mml:msub><mml:mi>w</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:msup><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:msubsup><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mfrac><mml:mo>,</mml:mo></mml:math><label>(2)</label></disp-formula>
<p>Where <inline-formula><mml:math id="M8"><mml:mspace width="0.1em"/><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the one-hot encoded ground-truth label at pixel position <inline-formula><mml:math id="M9"><mml:mi>i</mml:mi><mml:mspace width="0.25em"/></mml:math></inline-formula>for class <inline-formula><mml:math id="M10"><mml:mi>l</mml:mi></mml:math></inline-formula>.</p>
</sec>
<sec id="sec5">
<label>2.3</label>
<title>Automatic mask generator (AMG) hyperparameter optimization</title>
<p>Both SAM 1 and SAM 2 feature an AMG, which proposes masks without requiring a specific prompt input. Instead, a point grid is prompted internally, and predicted masks are filtered based on different tunable hyperparameters (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>; <xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>). Both the density of the point grid and the strictness of mask filtering can be set via the hyperparameters. Definitions of the hyperparameters can be found in the docstrings of the &#x201C;AutomaticMaskGenerator&#x201D; classes within the SAM 1 and SAM 2 Python packages.</p>
<p>To optimize the AMG hyperparameters per encoder, a grid search over the given sets of hyperparameter variations was conducted on 10 previously annotated maize ear images, which show, other than images of the MED, only the maize ear center (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure S1</xref>). As can be seen in <xref ref-type="table" rid="tab1">Table 1</xref>. The hyperparameter search space covered three different values for six hyperparameters, which covers 729 possible configurations in total. Some combinations did not produce any masks and even led to crashes of the SAM 2 package in 8 instances, which is why only 721 combinations are reported in this study. The specific faulty hyperparameter configurations can be seen in the ARAMSAM repository under the &#x201C;preprint_v0.1&#x201D; tag.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Hyperparameter search space for automatic mask generators (AMG) of SAM 1 (ViT-H) and SAM 2.1 (Hiera-S).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="middle" rowspan="2">Hyperparameter</th>
<th align="center" valign="middle" colspan="2">Values</th>
</tr>
<tr>
<th align="center" valign="bottom">SAM 1 (ViT-H)</th>
<th align="center" valign="top">SAM 2.1 (Hiera-S)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">points_per_side</td>
<td align="center" valign="bottom">{<italic>32</italic>, 64, <bold>128</bold>}</td>
<td align="center" valign="top">{<bold>
<italic>32</italic>
</bold>, 64, 128}</td>
</tr>
<tr>
<td align="left" valign="bottom">points_per_batch</td>
<td align="center" valign="bottom">{128}</td>
<td align="center" valign="top">{128}</td>
</tr>
<tr>
<td align="left" valign="bottom">pred_iou_thresh</td>
<td align="center" valign="bottom">{0.72, <bold>0.8</bold>, <italic>0.88</italic>}</td>
<td align="center" valign="top">{<bold>0.72</bold>, <italic>0.8</italic>, 0.88}</td>
</tr>
<tr>
<td align="left" valign="bottom">stability_score_thresh</td>
<td align="center" valign="bottom">{0.92, <italic>0.95</italic>, <bold>0.98</bold>}</td>
<td align="center" valign="top">{<bold>0.92</bold>, <italic>0.95</italic>, 0.98}</td>
</tr>
<tr>
<td align="left" valign="bottom">stability_score_offset</td>
<td align="center" valign="bottom">{0.7, <bold>
<italic>1.0</italic>
</bold>, 1.3}</td>
<td align="center" valign="top">{<bold>0.7</bold>, <italic>1.0</italic>, 1.3}</td>
</tr>
<tr>
<td align="left" valign="bottom">crop_n_layers</td>
<td align="center" valign="bottom">{<bold>
<italic>0</italic>
</bold>, 1, 2}</td>
<td align="center" valign="top">{<italic>0</italic>, <bold>1</bold>, 2}</td>
</tr>
<tr>
<td align="left" valign="bottom">crop_n_points_downscale_factor</td>
<td align="center" valign="bottom">{<bold>
<italic>1</italic>
</bold>, 2, 4}</td>
<td align="center" valign="top">{<bold>
<italic>1</italic>
</bold>, 2, 4}</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Italic values mark default values. Bold values mark optimal values.</p>
</table-wrap-foot>
</table-wrap>
<p>The chosen search space aims to increase the number of proposed masks compared to the default configuration. At the same time, it also explores hyperparameter values close to the default configuration. Hyperparameter settings that were considered less significant were not tested but were kept at default values and are not listed in <xref ref-type="table" rid="tab1">Table 1</xref>.</p>
<p>The <italic>F<sub>&#x03B2;</sub></italic>-score with <italic>&#x03B2;</italic>&#x202F;=&#x202F;2, weighing recall <italic>R</italic> four times as high as precision <italic>P</italic>, was chosen as a metric. Thereby, the production of more masks has been encouraged. The goal to increase the number of masks proposed by the AMG was driven by the assumption that manually discarding masks is less time-consuming than creating new masks manually for a human annotator. For precision and recall calculations, predicted and ground-truth masks were matched based on the intersection over union (<italic>IoU</italic>).</p>
<p>The <italic>IoU</italic> is defined as:</p>
<disp-formula id="E3"><mml:math id="M11"><mml:mi mathvariant="italic">IoU</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mtext>Area of Intersection</mml:mtext><mml:mtext>Area of Union</mml:mtext></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>&#x2223;</mml:mo><mml:mi>A</mml:mi><mml:mo>&#x2229;</mml:mo><mml:mi>B</mml:mi><mml:mo>&#x2223;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x2223;</mml:mo><mml:mi>A</mml:mi><mml:mo>&#x222A;</mml:mo><mml:mi>B</mml:mi><mml:mo>&#x2223;</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:math><label>(3)</label></disp-formula>
<p>Where the area of intersection is the overlapping region of two masks, and the area of union is the area covered by both masks combined.</p>
<p>Since predicted masks are not directly used for a downstream task but instead are used as annotations, true positives (TP) are defined at the ground-truth level. A ground-truth mask is counted as a TP if there exists at least one predicted mask with an Intersection-over-Union (IoU) greater than 0.8. This threshold was selected empirically based on preliminary tests that demonstrated sufficient mask quality. If a ground-truth mask is not matched with any predicted mask, it is counted as FN. False positives (FP) are predicted masks that cannot be associated with any ground-truth mask above the IoU threshold. Consequently, these definitions do not follow conventional one-to-one matching between predictions and ground truths. This is intentional since the annotation pipeline, in theory, allows a single predicted mask to be reused for multiple ground-truth instances, though this is very rare. Such a scenario would be a ground-truth instance that is occluded by another ground-truth instance. Here, the same predicted mask could be suitable to represent both the occluded instance and the instance on top. Thus, a single prediction could represent multiple TP.</p>
<p>Precision <italic>P</italic> is defined as:</p>
<disp-formula id="E4"><mml:math id="M12"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mi mathvariant="italic">TP</mml:mi><mml:mrow><mml:mi mathvariant="italic">TP</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="italic">FP</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:math><label>(4)</label></disp-formula>
<p>while recall <italic>R</italic> is defined as:</p>
<disp-formula id="E5"><mml:math id="M13"><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mi mathvariant="italic">TP</mml:mi><mml:mrow><mml:mi mathvariant="italic">TP</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="italic">FN</mml:mi></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:math><label>(5)</label></disp-formula>
<p>Thus, the <italic>F<sub>&#x03B2;&#x202F;=&#x202F;2</sub></italic>-score is defined as:</p>
<disp-formula id="E6"><mml:math id="M14"><mml:msub><mml:mi>F</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:mo stretchy="true">(</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:msup><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="true">)</mml:mo><mml:mspace width="0.1em"/><mml:mfrac><mml:mrow><mml:mi>P</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msup><mml:mspace width="0.1em"/><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>R</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mn>5</mml:mn><mml:mspace width="0.1em"/><mml:mfrac><mml:mrow><mml:mi>P</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>4</mml:mn><mml:mspace width="0.1em"/><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>R</mml:mi></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:math><label>(6)</label></disp-formula>
</sec>
<sec id="sec6">
<label>2.4</label>
<title>ARAMSAM software</title>
<p>ARAMSAM is a previously unpublished open-source image annotation software, developed in this study for instance segmentation and mask transfer from one overlapping image to the next. The software uses Python (3.10) (<xref ref-type="bibr" rid="ref40">Van Rossum and Drake, 2009</xref>) and is based on publicly available packages of the Python universe. The software&#x2019;s front end runs on PyQt6 (6.7) (<xref ref-type="bibr" rid="ref47">Riverbank Computing, 2025</xref>) while the back end uses OpenCV (4.10) (<xref ref-type="bibr" rid="ref3">Bradski, 2000</xref>) for conventional computer vision tasks, Pandas for data wrangling (2.2) (<xref ref-type="bibr" rid="ref19">McKinney, 2010</xref>), PyTorch (2.4) (<xref ref-type="bibr" rid="ref26">Paszke et al., 2019</xref>) for AI utilities and SAM 1 (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>) and SAM 2 (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>) for semiautomatic zero-shot segmentation tasks.</p>
<p>The user interface of ARAMSAM features a top bar with general settings and buttons for annotation actions (<xref ref-type="fig" rid="fig2">Figure 2</xref>). Below the bar, four freely selectable views of the image that is being annotated are visible. To ensure a comparable annotation process during the experiments, users were not allowed to change any settings themselves, and the views were predefined. The top-left view showed the original RGB image, and the top-right view showed the collection of previously annotated masks on the original. The bottom-right view showed the previously annotated masks in white, with overlapping mask parts being highlighted in red to avoid unintended overlap. The bottom-left view indicated points for both drawing polygons and interactively prompting with SAM 1 or SAM 2.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Overview of the ARAMSAM user interface. Top-left: Original RGB image. Top-right: Annotated masks overlaid on the original RGB image, with numbers indicating the mask ID. The white mask represents the preview generated by SAM 1/SAM 2. Bottom-left: Point view showing annotation prompts. Red points indicate negative prompts guiding SAM 1/SAM 2 to avoid these locations. White points indicate positive prompts guiding SAM 1/SAM 2 to include these locations. The preview mask is shown in green. Bottom-right: Annotated masks on black background. Overlapping mask areas are highlighted in red and the preview mask is shown in blue.</p>
</caption>
<graphic xlink:href="frai-08-1748468-g002.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Close-up views of corn kernels on a cob, divided into four quadrants. The top left shows natural corn kernels. The top right features the kernels with colored segmentation and numbers. The bottom left highlights a selected kernel in green. The bottom right displays a black and white segmentation mask of the kernels.</alt-text>
</graphic>
</fig>
<p>ARAMSAM enables creating segmentation masks with a set of tools to meet different demands for specific applications. The most common approach is to draw polygons around the object of interest. Another tool is the interactive prompting based on SAM 1 or SAM 2. Here, the user hovers the mouse over the image for real-time mask proposals (<xref ref-type="fig" rid="fig2">Figure 2</xref>). The image is embedded by the encoder beforehand. Multiple positive and negative points can be added to refine the proposed mask. Additionally, ARAMSAM employs AMG as a supplementary tool within SAM 1 and SAM 2. Thereby, masks are proposed sequentially, and the user&#x2019;s task is to choose whether each mask represents an object of interest or should be discarded instead.</p>
<p>Furthermore, ARAMSAM includes functionalities to transfer masks from one image to another if the dataset contains consecutive, overlapping images. Since SAM 1 does not inherently feature mask propagation, a panorama-based algorithm is used to transfer masks. Here, image key points are detected by the ORB (oriented FAST and rotated BRIEF) feature detector (<xref ref-type="bibr" rid="ref32">Rublee et al., 2011</xref>), which is based on FAST (features from accelerated segment test) (<xref ref-type="bibr" rid="ref31">Rosten and Drummond, 2006</xref>) and BRIEF (binary robust independent elementary features) (<xref ref-type="bibr" rid="ref5">Calonder et al., 2010</xref>). The key points are matched as shown by <xref ref-type="bibr" rid="ref4">Brown and Lowe (2007)</xref>. The resulting image orientation can be exploited to project bounding boxes of objects annotated on the first image to the following image. The projected bounding boxes are then prompted to SAM 1 with the second image. When using SAM 2 in ARAMSAM, masks are propagated by means of the mask propagation functionalities, which were originally designed for video object segmentation (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>). To load image sequences instead of video frames into the memory bank of SAM 2, a custom function was added to the original Python package.</p>
</sec>
<sec id="sec7">
<label>2.5</label>
<title>User experiment</title>
<p>Fourteen experts in the field of agriculture were asked to annotate images of three randomly selected maize ears from MED in the ARAMSAM user interface. To familiarize participants with the software, each individual completed a tutorial demonstrating how to identify valid kernel masks and how to use all relevant tools for the experiment. During the experiment, every participant applied three different annotation methods to the same three initially selected maize ears. To mitigate potential learning effects over time, the order of these nine method&#x2013;ear combinations was randomized for each user. An overview of the annotation methods is provided in <xref ref-type="table" rid="tab2">Table 2</xref>.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Overview of annotation methods in the user experiment.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Annotation method</th>
<th align="left" valign="top">Tools</th>
<th align="center" valign="top">Number of ears</th>
<th align="center" valign="top">Instance limit</th>
<th align="center" valign="top">Images per ear</th>
<th align="center" valign="top">Mask transfer</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Polygon</td>
<td align="left" valign="middle">Polygon annotation of highlighted maize kernels</td>
<td align="center" valign="middle" rowspan="3">3</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">1</td>
<td align="left" valign="middle">&#x2013;</td>
</tr>
<tr>
<td align="left" valign="middle">SAM 1</td>
<td align="left" valign="middle" rowspan="2">1. Select AMG masks (image 1)2. Interactive prompting (image 1)3. Polygon drawing (image 1)4. Mask transfer and manual control (from image 1 to 2)5. Select AMG masks (image 2)6. Interactive prompting (image 2)7. Polygon drawing (image 2)</td>
<td align="center" valign="middle" rowspan="2">&#x2013;</td>
<td align="center" valign="middle" rowspan="2">2</td>
<td align="left" valign="middle">Panorama matching</td>
</tr>
<tr>
<td align="left" valign="middle">SAM 2</td>
<td align="left" valign="middle">Mask propagation</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The users had to apply the three different annotation methods in separate steps of the experiment. SAM 1 and SAM 2 represent annotation methods consisting of multiple sequentially applied tools based on SAM 1 (Vit-H) and SAM 2.1 (Hiera-S), respectively. AMG: Automatic mask generator.</p>
</table-wrap-foot>
</table-wrap>
<p>When participants were asked to use the polygon method, they were only required to annotate three maize kernels to avoid excessive effort. Before the experiment, three kernels per image were randomly selected and highlighted by bounding boxes, ensuring that all users annotated the same kernels.</p>
<p>For the annotation methods based on SAM 1 and SAM 2, users were given a fixed structure, as listed in the tool&#x2019;s column of <xref ref-type="table" rid="tab2">Table 2</xref>. These annotation methods with a fixed structure had to be applied to three image pairs that contain two consecutive images. The three image pairs were the same for SAM 1 and SAM 2, and the first image of each of the image pairs is used during the polygon method. The fixed structure is ordered from tasks requiring less interaction (e.g., selecting AMG masks) to tasks requiring more interaction (e.g., polygon drawing). If, e.g., all valid maize kernels of one image had been assigned a good mask created by the AMG, there was no need to apply interactive prompting or polygon drawing. After transferring masks from the first image to the second (either by the panorama approach or the SAM 2 propagation), the users were asked to check whether all masks had been transferred correctly and to remove invalid masks by clicking on them. The criteria for a mask being transferred correctly are assessed only by the quality of the mask on the second image. Individual maize kernels are not tracked back to the preceding image.</p>
<p>To evaluate whether annotation methods based on SAM 1 and SAM 2 can accelerate the annotation process of instance segmentation masks over previous standard methods, the drawing of polygons was used as a baseline. Since the annotation of the second image of an ear is influenced by the mask transfer capabilities of both the SAM 1 and the SAM 2 method, only the first image of an ear was taken for comparison across the polygon and both SAM methods. When the users were applying annotation methods based on SAM 1 and SAM 2, the users had to independently decide which object represented a valid maize kernel. To study the consistency of annotation decisions across different users, the annotation frequency per image pixel <italic>f<sub>a,px</sub></italic> was defined as:</p>
<disp-formula id="E7"><mml:math id="M15"><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mi mathvariant="italic">px</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>N</mml:mi><mml:mtext>assigned</mml:mtext></mml:msub><mml:msub><mml:mi>N</mml:mi><mml:mtext>rounds</mml:mtext></mml:msub></mml:mfrac><mml:mo>,</mml:mo></mml:math><label>(7)</label></disp-formula>
<p>Where <italic>N</italic><sub>assigned</sub> is the number of times a pixel has been assigned to a mask; <italic>N</italic><sub>rounds</sub> is the number of annotation rounds per image. With 14 users, each employing two annotation methods based on SAM 1 and SAM 2, the total number of annotation rounds per image was <italic>N</italic><sub>rounds</sub>&#x202F;=&#x202F;28.</p>
<p>All experiments have been conducted on systems using a single RTX 3090 (NVIDIA Corporation, Santa Clara, USA) as a GPU.</p>
</sec>
<sec id="sec8">
<label>2.6</label>
<title>Statistical analysis and data visualization</title>
<p>All statistical analyses were performed using R (4.3.2) (<xref ref-type="bibr" rid="ref39">Team, 2025</xref>). Data wrangling and manipulation were carried out with Dplyr (1.1), Tidyr (1.3), and Tibble (3.2) from the Tidyverse universe (<xref ref-type="bibr" rid="ref45">Wickham et al., 2019</xref>). Statistical tests and <italic>post hoc</italic> analyses were conducted using Rstatix (0.7) (<xref ref-type="bibr" rid="ref14">Kassambara, 2019</xref>). Normalized metric result data [<italic>p</italic> &#x2208; (0,1)] was pre-processed with a logit transformation before applying statistical tests. Thereby, boundary constraints close to 0 or 1 and variance heterogeneity were tackled as shown in <xref ref-type="bibr" rid="ref50">Zou et al. (2004)</xref>. Repeated measures ANOVA results have been corrected by the Greenhouse&#x2013;Geisser approach to mitigate sphericity of within-subject factors. Two-sided pairwise <italic>t</italic>-tests with Bonferroni correction have been conducted as post-hoc tests. Both the repeated measures ANOVA and the post-hoc tests rejected the H<sub>0</sub>-Hypothesis with a significance level of <italic>&#x03B1;</italic>&#x202F;=&#x202F;0.05.</p>
<p>Python (3.10) (<xref ref-type="bibr" rid="ref40">Van Rossum and Drake, 2009</xref>) and the Pandas package (2.2) (<xref ref-type="bibr" rid="ref19">McKinney, 2010</xref>) were used for data preparation, followed by data visualization based on Matplotlib (3.10) (<xref ref-type="bibr" rid="ref13">Hunter, 2007</xref>) and Seaborn (0.13) (<xref ref-type="bibr" rid="ref44">Waskom et al., 2021</xref>). In all boxplots displayed in this study, the central box spans from the first quartile to the third quartile with a line inside marking the median. The whiskers extend to the smallest and largest data points within 1.5 times the interquartile range from the quartiles. Data points falling outside these limits are plotted individually as outliers.</p>
</sec>
<sec id="sec9">
<label>2.7</label>
<title>Declaration of generative AI and AI-assisted technologies in the writing process</title>
<p>During the preparation of this study, the authors used ChatGPT 4-o (OpenAI, Inc., San Francisco, USA) to improve the writing. After using this tool/service, the authors reviewed and edited the content as needed and take full responsibility for the published article.</p>
</sec>
</sec>
<sec sec-type="results" id="sec10">
<label>3</label>
<title>Results</title>
<sec id="sec11">
<label>3.1</label>
<title>Zero-shot performance of different SAM 1 and SAM 2 encoders</title>
<p>To evaluate mask quality as predicted by SAM 1 and SAM 2, all encoders have been applied on the agricultural datasets (MED, MUD, SOD), with a single point (the geometric median) for each of the 100 ground-truth masks&#x2014;a comparison between the predicted mask and the ground-truth mask results in the <italic>GDS</italic>. In <xref ref-type="fig" rid="fig3">Figure 3</xref>, the resulting <italic>GDS</italic> of the encoder experiments are displayed for the MED, MUD and SOD dataset (from top to bottom). The scores range from 0 to 1 and are displayed as a boxplot indicating the distribution of the quartiles. All encoders achieve relatively high mean <italic>GDS</italic> for both the MED (0.87) and the SOD (0.94) compared to the MUD (0.50). These results were expected as the ground-truth masks of both the MED and SOD resemble compact round objects, whereas the plant objects in the MUD are complex and partially overlap with neighboring plants.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Segmentation quality of SAM 1 and SAM 2 across different encoder versions tested on the maize ear dataset (MED), maize UAV dataset (MUD), and soil dataset (SOD). The <italic>x</italic>-axis indicates the encoder versions, and the <italic>y</italic>-axis shows the generalized dice score (<italic>GDS</italic>) per mask. Significant differences are denoted by letters arranged from the highest mean <italic>GDS</italic> to the lowest (<italic>&#x03B1;</italic>&#x202F;&#x003C;&#x202F;0.05).</p>
</caption>
<graphic xlink:href="frai-08-1748468-g003.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Box plots show the GDS per mask for MED, MUD, and SOD across different SAM versions. The versions, labeled as ViT-B, ViT-L, ViT-H, Hiera-T, Hiera-S, Hiera-B+, and Hiera-L, are color-coded: purple for version 1, dark green for version 2.0, and teal for version 2.1. Each plot shows median, quartiles, and outliers, labeled with letters indicating statistical grouping.</alt-text>
</graphic>
</fig>
<p>To compare the results of all individual encoders, a repeated measures ANOVA was conducted per dataset. Significant effects across encoders were revealed for all datasets, with the test results being <italic>F</italic>(2.96, 292.92)&#x202F;=&#x202F;165.48, <italic>p</italic>&#x202F;&#x003C;&#x202F;0.001, for the MED, <italic>F</italic>(6.23, 616.59) =&#x202F;17.60, <italic>p</italic> &#x003C;&#x202F;0.001, for the MUD, and <italic>F</italic>(4.44, 439.61) =&#x202F;7.71, <italic>p</italic> &#x003C;&#x202F;0.001, for the SOD. A pairwise <italic>t</italic>-test was used as a post-hoc test (<italic>&#x03B1;</italic>&#x202F;&#x003C;&#x202F;0.05) with significant differences shown in <xref ref-type="fig" rid="fig3">Figure 3</xref>. Two encoders not sharing a letter achieved significantly different performance on the respective dataset. The alphabetical order of the letters indicates the performance ranking from &#x201C;a&#x201D; the best to &#x201C;g&#x201D; the worst performing group.</p>
<p>The Hiera-T encoder of SAM 2.0, as well as both versions (SAM 2.0 and SAM2.1) of the Hiera-L encoder, had to be excluded from statistical tests of the MED since normality of the data could not be assumed. This is indicated by the absence of a letter. Since these three encoders apparently do not show good performance, they are irrelevant for further experiments and thus can be excluded from statistical tests. This performance test of the encoders was conducted to identify the best encoders for later experiments.</p>
<p>Notably, at least one encoder of SAM 1 belongs to the significant letter &#x201C;a&#x201D; for all three datasets. Moreover, all encoders of SAM 1 show significantly higher <italic>GDS</italic> per mask than any encoder of SAM 2 when tested on MED, indicating stronger or equal performance of SAM 1 on all datasets when compared to SAM 2.</p>
<p>The compared encoders vary regarding their network architecture and the number of parameters (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>; <xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>). Thus, the largest ViT-H (SAM 1) and Hiera-L (SAM 2) encoders are less computation-efficient than their smaller counterparts. <xref ref-type="fig" rid="fig3">Figure 3</xref> indicates that the smallest encoder types of both SAM 1 (ViT-B) and SAM 2 (Hiera-T) show significantly lower <italic>GDS</italic> than the larger encoders. However, the largest encoder types (ViT-H, Hiera-L) do not perform significantly better than the medium-sized encoders. Interestingly, the updated encoders of SAM 2.1 only show a significantly higher <italic>GDS</italic> than their predecessors for the Hiera-S encoder at MED and for the Hiera-L encoder at the SOD. In contrast, the old SAM 2.0 encoders achieved a significantly higher <italic>GDS</italic> for the Hiera-B+ encoder at MED and the Hiera-S encoder at the SOD. These results indicate no improvement of the updated SAM 2.1 encoders over the initially released ones on the proposed agricultural use cases.</p>
</sec>
<sec id="sec12">
<label>3.2</label>
<title>Automatic mask generator (AMG) hyperparameter optimization</title>
<p>The following experiments focused on images of the MED because maize ears are complex, round objects that were captured from different angles. Thus, object tracking was expected to be a more challenging task than with images of the MUD and SOD where all image planes are parallel to another and to the soil surface. Moreover, only the best performing encoders of SAM 1 (ViT-H) and SAM 2.1 (Hiera-S) according to <italic>GDS</italic>, as achieved on the MED, were selected for optimizing hyperparameters of the AMG. The best performing hyperparameters according to the mean <italic>F<sub>2</sub></italic> over all images have been identified on a new subset of the MED, where ground-truth masks for all kernels have been annotated manually (<xref ref-type="table" rid="tab1">Table 1</xref>).</p>
<p>Within the optimal settings of SAM 1 (ViT-H), the identified values for the hyperparameters <italic>points_per_side</italic> and <italic>pred_iou_thresh</italic> are increasing the number of proposed masks compared to the default values but in contrast, the setting of <italic>stability_score_thresh</italic> is applying a stronger filter to the proposed masks than the default value. However, the settings of hyperparameters for SAM 2.1 (Hiera-S) increases the number of masks by a reduced value of <italic>pred_iou_thresh</italic>, by an increased number of <italic>crop_n_layers</italic> and with reduced <italic>stability_score_thresh</italic> as well as reduced <italic>stability_score_offset</italic>. Notably, the optimal hyperparameters of SAM 2.1 (Hiera-S) include no deviation from the default settings that would reduce the number of masks. These different optimal settings highlight the models&#x2019; network architectural differences as revealed on the MED.</p>
<p><xref ref-type="fig" rid="fig4">Figure 4</xref> shows the <italic>F<sub>2</sub></italic>-score of the selected SAM 1 encoder (ViT-H) and SAM 2.1 encoder (Hiera-S) per test image. A significant effect of the encoders and hyperparameters was revealed by repeated measures ANOVA (<italic>F</italic>(1.04, 9.4)&#x202F;=&#x202F;31.27, <italic>p</italic>&#x202F;&#x003C;&#x202F;0.001). The AMG of both SAM benefits significantly from the hyperparameter optimization compared to the default configuration. While SAM 1 (ViT-H) improved from a mean <italic>F<sub>2</sub></italic>-score of 0.87&#x2013;0.93, SAM 2.1 (Hiera-S) improved from 0.05 to 0.74.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Mask quality predicted by automatic mask generators (AMG) before (default) and after hyperparameter optimization (best). The models were evaluated per image using F2 calculated from the number of predicted masks that matched ground-truth masks. Significant differences (pairwise <italic>t</italic>-test, <italic>&#x03B1;</italic>&#x202F;&#x003C;&#x202F;0.05) are indicated by different letters.</p>
</caption>
<graphic xlink:href="frai-08-1748468-g004.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Box plot comparing F-score per image for SAM 1 ViT-H and SAM 2.1 Hiera-S models with different hyperparameters. SAM 1 ViT-H shows higher scores for both default (blue) and best (orange) hyperparameters compared to SAM 2.1 Hiera-S.</alt-text>
</graphic>
</fig>
<p>Again, the encoder representing SAM 1 (ViT-H) performs significantly better than SAM 2.1 (Hiera-S). Moreover, the SAM 2.1 (Hiera-S) encoder achieved an especially low <italic>F<sub>2</sub></italic> on one image, depicted as an outlier in <xref ref-type="fig" rid="fig4">Figure 4</xref>. Conversely, the selected SAM 1 (ViT-H) encoder does not show any outliers, indicating more stable performance on the MED.</p>
</sec>
<sec id="sec13">
<label>3.3</label>
<title>User experiment</title>
<p>After identifying the best-performing encoders and the optimal AMG hyperparameters for SAM 1 (ViT-H) and SAM 2 (SAM 2.1 Hiera-S), respectively, a user experiment based on ARAMSAM was conducted with these configurations. All users had to apply annotation methods based on SAM 1, SAM 2, and the drawing of polygons. Thus, each image has been annotated by each user in multiple rounds. While, the objects of interest were highlighted during the polygon method, when using the other methods, the participants had to decide on their own which objects represent valid kernels according to the instructions they were given in the tutorial. The annotation decisions for all three image pairs are shown in <xref ref-type="fig" rid="fig5">Figure 5</xref>. As shown in the bottom row of <xref ref-type="fig" rid="fig5">Figure 5</xref>, the number of annotated instances (maize kernels) varied little across different users. The highest standard deviation of annotated instances per image (3.9 kernels) is observed for the first image of the left image pair. Consistent annotations across users are also confirmed by the top row of <xref ref-type="fig" rid="fig5">Figure 5</xref>, where most kernels have been annotated at annotation frequency <italic>f<sub>a, px</sub></italic> close to 1.0. Nevertheless, lower <italic>f<sub>a, px</sub></italic> can be observed for kernels in the area of the infertile tip, as shown in the left and the center image pairs. An enlarged view of the leftmost image is displayed in the supplementary data (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure S2</xref>). The rare occurrence of pink color on some kernel edges indicates that overlapping masks were assigned to neighboring kernels, which represents under-segmentation. Conversely, the light-blue color on kernel edges indicates that the assigned kernel masks were too small and did not cover the entire kernel. This over-segmentation can be observed on all ear images (<xref ref-type="fig" rid="fig5">Figure 5</xref>). Yet, both under-segmentation and over-segmentation usually cover a few pixels, which should have a minor influence on applications such as phenotyping.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Annotation decisions of 14 users for three maize image pairs. Top: Relative frequency a pixel has been assigned to a mask (<italic>f<sub>a, px</sub></italic>) for the total number of annotation rounds for the first (<italic>&#x03C9;</italic>&#x202F;=&#x202F;0&#x00B0;) and second rotated image (<italic>&#x03C9;</italic>&#x202F;=&#x202F;7.1&#x00B0;) of a maize ear. Only pixels assigned to a mask more than once are included. Bottom: Number of kernel instances annotated per image over all applied methods. The polygon method is excluded from the figure.</p>
</caption>
<graphic xlink:href="frai-08-1748468-g005.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Top row shows three pairs of corn cobs with color-coded kernels. The color scale on the left represents kernel values from zero to two. Bottom row displays four box plots labeled "First" and "Second," indicating annotated instances on the y-axis and image positions on the x-axis.</alt-text>
</graphic>
</fig>
<p>In <xref ref-type="fig" rid="fig6">Figure 6</xref>, the annotation time per mask is displayed for the SAM 1 and SAM 2 approaches for both the first and the second of the consecutive images, as well as for the polygon method for the first of the consecutive images. Each boxplot contains 42 data points (14 users times 3 images). A significant effect of the annotation method on the annotation time per mask was revealed by a repeated measures ANOVA (<italic>F</italic>(1.11, 14.43)&#x202F;=&#x202F;64.70, <italic>p</italic>&#x202F;&#x003C;&#x202F;0.001). The subsequent post-hoc test shows significant differences between the polygon method and both SAM methods (indicated by different letters). Accordingly, a significant difference in annotation time was observed with 9.7&#x202F;s/mask for the polygon method. The approaches based on SAM 1 and SAM 2 took 2.1 and 2.6&#x202F;s/mask, respectively. However, no significant differences between SAM 1 and SAM 2 were observed in the first images.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Annotation time per mask across different methods (X-axis). Significant differences are indicated by lowercase letters across first images and by capital letters across second images (pairwise <italic>t</italic>-test, <italic>&#x03B1;</italic>&#x202F;&#x003C;&#x202F;0.05).</p>
</caption>
<graphic xlink:href="frai-08-1748468-g006.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Box plot comparing annotation time in seconds per mask for three methods: Polygon, SAM 1, and SAM 2. Polygon shows the highest median time. Colored boxes represent two positions: yellow for first and orange for second. Data presented includes outliers.</alt-text>
</graphic>
</fig>
<p>In the next step, masks from the first images were transferred to the second images either by the panorama-based algorithm (SAM 1) or mask propagation (SAM 2). On the second images a significant difference between SAM 1 and SAM 2 could be shown by both the ANOVA (<italic>F</italic>(1, 13)&#x202F;=&#x202F;24.177, <italic>p</italic> &#x003C;&#x202F;0.001) and the post-hoc test. The latter indicated a significantly lower annotation time of SAM 2 when compared to SAM 1 (<xref ref-type="fig" rid="fig6">Figure 6</xref>). Strikingly, the mean of SAM 1 on the second image (3.3&#x202F;s/mask) is higher than that on the first image (2.1&#x202F;s/mask).</p>
<p><xref ref-type="fig" rid="fig7">Figure 7</xref> depicts the number of masks generated and the annotation time (s/mask) for each tool. The AMG was the most applied tool of the SAM 1 method on the second image (56.0%) (<xref ref-type="fig" rid="fig7">Figure 7</xref>), suggesting that the majority of masks were not transferred correctly from the first to the second image. Likewise, the AMG of SAM 1 and the AMG of SAM 2 were also the predominant tools on the first images, where users could not benefit from transferred masks. Here, AMG was the origin of 95.8% (SAM 1) and 94.9% (SAM 2) of the selected masks (<xref ref-type="fig" rid="fig7">Figure 7</xref>). Thus, the users&#x2019; annotation behavior when transferring a mask from panorama matching (SAM 1) was similar to starting from a new image. This shows that SAM 1 did not benefit from the previous annotations and the panorama-inspired method for mask transfer of SAM 1 appears to not be suitable for the MED.</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Applied tools within annotation methods. Data points represent one annotation round per user. Time (s/mask) is based on <italic>the</italic> mean temporal distance between individual masks when more than one mask was annotated per tool. AMG: Automatic Mask Generator.</p>
</caption>
<graphic xlink:href="frai-08-1748468-g007.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Box plots comparing two methods, SAM 1 and SAM 2, in four scenarios: AMG, Interactive, Polygon, and Mask transfer, across two images. The top plots show the number of masks, while the bottom plots show annotation time per mask in seconds. Both methods generally display similar performance, with notable variations in the Polygon scenario. SAM 2 slightly outperforms SAM 1 in the second image's Polygon scenario in terms of time.</alt-text>
</graphic>
</fig>
<p>However, the SAM 2 method on the second image represents the fastest method over both image positions by requiring 1.6&#x202F;s/mask on average. This is also supported by mask transfer being the predominant origin of masks created by the SAM 2 method on the second image (94.0%) (<xref ref-type="fig" rid="fig7">Figure 7</xref>). To determine whether the SAM 2 method, benefiting from masks propagated from the previous image, allows significantly faster annotation time than applying SAM 1 directly on the first image, a one-sided pairwise <italic>t</italic>-test was conducted. Here, the time per mask has been averaged over the images. The test results show significantly faster annotation times of SAM 2 (<italic>t</italic>(13)&#x202F;=&#x202F;2.03, <italic>p</italic>&#x202F;=&#x202F;0.032), suggesting that applying SAM 2 with mask propagation on image sequences of the MED is the fastest of the proposed methods.</p>
</sec>
</sec>
<sec sec-type="discussion" id="sec14">
<label>4</label>
<title>Discussion</title>
<sec id="sec15">
<label>4.1</label>
<title>Comparing zero-shot performance of SAM1 and SAM2</title>
<p>Throughout this study, SAM 1 and SAM 2 were compared according to mask quality per prompt on three datasets (MED, MUD, SOD), mask coverage of the respective AMG on one dataset (MED), and temporal annotation effort for users on one dataset (MED). SAM 2 did not outperform its predecessor in any of these disciplines when applied to single images. Architecture-specific problems became apparent during the encoder experiment, where both Hiera-L encoders and the initial Hiera-T (2.0) encoder of SAM 2 are predicting the entire ear as a single mask instead of individual kernels in multiple instances (<xref ref-type="fig" rid="fig3">Figure 3</xref>). These problems specific to a certain encoder size highlight the need for proper model selection depending on the data.</p>
<p>While this study covered only agricultural use cases with RGB data, <xref ref-type="bibr" rid="ref34">Sengupta et al. (2025)</xref>, compared SAM 1 and SAM 2 on medical datasets covering both RGB and grayscale data. The authors showed that SAM 2 does not consistently perform better than SAM 1, which appears to be independent of image data type. However, SAM 2 achieved higher mask accuracy metrics in segmenting solar panels on remote sensing data (<xref ref-type="bibr" rid="ref28">Rafaeli et al., 2024</xref>), in contrast to this study especially SAM 2.1 (Hiera-L) outperformed SAM 1 (ViT-L). <xref ref-type="bibr" rid="ref29">Ravi et al. (2024)</xref> showed an improvement of SAM 2 for zero-shot performance (single images) on the most of 37 datasets from multiple domains. However, the 37 datasets barely focus on agriculture, besides the PPDLS (plant phenotyping datasets leaf segmentation) (<xref ref-type="bibr" rid="ref20">Minervini et al., 2016</xref>), containing plant phenotyping data. Here, the performance of SAM 2 showed a delta of &#x2212;4.8 mIoU (mean intersection over union) compared to the performance of SAM 1 (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>), indicating a setback in the performance of SAM 2 over its predecessor. The findings from <xref ref-type="bibr" rid="ref29">Ravi et al. (2024)</xref> on the PPDLS, together with the results presented here, suggest that SAM 2 does not represent an improvement over its predecessor regarding zero-shot performance on single images of agricultural datasets.</p>
<p>Yet, <xref ref-type="bibr" rid="ref29">Ravi et al. (2024)</xref> and <xref ref-type="bibr" rid="ref28">Rafaeli et al. (2024)</xref> showed better performance of SAM 2 compared to SAM 1 in most domains. This is especially noteworthy because of SAM 2 enormous time savings in computation. Due to the smaller hierarchical image encoders, SAM 2 is six times faster than SAM 1 (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>). The architectural differences could be considered as a reason why the optimal parameters of the selected encoders of SAM 1 and SAM 2 differ greatly. Another benefit of SAM 2 is its ability to propagate masks from one image frame to the next. Incorporated in ARAMSAM, this feature accelerated the annotation time from 2.1&#x202F;s/mask to 1.6&#x202F;s/mask, by about 23% (<xref ref-type="fig" rid="fig6">Figure 6</xref>). Despite SAM 2 not showing improved mask quality compared to SAM 1, the mask propagation capability, as well as the more efficient model architecture, can lead to SAM 2 accelerating annotation times and saving human labor. Especially on image sequences, SAM 2 is considered the most suited method for annotation with ARAMSAM on the MED.</p>
<p>The positive results of our experiments are valid for the controlled environments represented by the datasets that all originated from the same geographical region. Transferring the results to less controlled conditions would expose the models to challenging properties like the tempo-spatial variations of environments, e.g., through weather effects or seasonal growth. These include background disturbances, partial or full occlusion of relevant object features or complete objects, different object sizes, rotations, or deformed objects, e.g., through wind and illumination changes due to varying daylight conditions (<xref ref-type="bibr" rid="ref37">Song et al., 2025</xref>). Furthermore, the process of image acquisition (sensor type, motion blur, processing algorithms) also affects the image quality, which is important for successful use with deep learning methods (<xref ref-type="bibr" rid="ref10">Dodge and Karam, 2016</xref>). Despite the limited variability of datasets in the experiments of this study, a certain robustness of ARAMSAM to new conditions would be expected, since SAM 1 and SAM 2 are foundation models that have been trained on large and diverse datasets (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>).</p>
</sec>
<sec id="sec16">
<label>4.2</label>
<title>Parameter optimization of automatic mask generator (AMG)</title>
<p>To our knowledge, this is the first study doing a hyperparameter optimization on the automatic mask generator of both SAM 1 and SAM 2. In the user experiment the AMG was the most used tool for both image positions of the SAM 1 method and the most used tool of the SAM 2 method on first images (<xref ref-type="fig" rid="fig7">Figure 7</xref>). Although it should be noted that the here proposed structured experiment design fostered the usage of the AMG being used by the participants in first or second position, the capability of the AMG to cover more than 95.8% (SAM 1) and 94.9% (SAM 2) of the valid maize kernels is remarkable. Of all tools, the AMG of both SAM 1 and SAM 2 showed the lowest annotation time on the first images, taking only taking 1.5&#x202F;s/mask and 1.7&#x202F;s/mask, respectively (<xref ref-type="fig" rid="fig7">Figure 7</xref>). At the same time, the data shows a low standard deviation of 0.7&#x202F;s/mask (SAM 1) and 0.6&#x202F;s/mask (SAM 2), manifesting the tools&#x2019; reliability. As demonstrated in <xref ref-type="fig" rid="fig4">Figure 4</xref>, the AMG of both SAM 1 and SAM 2 benefited greatly from the hyperparameter optimization. Especially the results of SAM 2, improving the <italic>F<sub>2</sub></italic> by more than 14 times, underline the importance of hyperparameter optimization when applying the AMG.</p>
<p>Conversely, the AMG could propose a large share of useless masks in scenarios where only a few objects of interest exist in one image. However, in crowded scenes where most objects represent object of interest like in the MED, the AMG can be especially useful. Therefore, exploiting the potential of this powerful tool by hyperparameter optimization is an important contribution to accelerating the annotation of segmentation datasets.</p>
</sec>
<sec id="sec17">
<label>4.3</label>
<title>Time savings by ARAMSAM orchestrating SAM-based annotation tools</title>
<p>Applying annotation tools based on both SAM 1 and SAM 2 clearly outperformed the polygon method representing a former state of the art method for annotation of segmentation masks. For single images, the annotation time per mask is accelerated by 4.6 times for SAM 1 and 3.7 times for SAM 2. When applying SAM 2 with mask propagation on image sequences, the acceleration increases by a factor of 6.1 compared to the polygon method. Yet, it should be noted that the panorama-based mask transfer of masks proposed by SAM 1 did slow down the annotations by factor 1.6 compared to applying SAM 1 without any mask transfer. This highlights the difficulties of mask transfer even on highly overlapping images and indicates that this method was not suitable for mask transfer on the MED. Likely, the panorama-based mask transfer would have performed better on image sequences moving in a linear direction instead of the circular rotation presented by the MED. A more computation-intensive alternative could be a structure from motion (<xref ref-type="bibr" rid="ref33">Schonberger and Frahm, 2016</xref>) based approach. Like the panorama algorithm, structure from motion matches multiple key points from overlapping images. In contrast to panorama stitching, the key points as well as the camera positions are oriented in 3D space, which would allow mask transfer even on irregularly shaped objects such as maize ears. However, structure from motion requires multiple images and sophisticated computation hardware to be applied in an edge scenario like image annotation (<xref ref-type="bibr" rid="ref33">Schonberger and Frahm, 2016</xref>).</p>
<p>The SAM tools implemented in ARAMSAM can save a tremendous amount of labor on the MED dataset compared to polygon drawing. Since SAM 1 and SAM 2 were trained and successfully tested on various domains (<xref ref-type="bibr" rid="ref16">Kirillov et al., 2023</xref>; <xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>), ARAMSAM has the potential to further accelerate the annotation process in other domains than the MED. Yet, it should be noted that the maize kernels represent rather simple objects with regular, round shapes and clear edges. While demonstrating how SAM 1 and SAM 2 orchestrated by ARAMSAM accelerate annotation speed, this study does not compare ARAMSAM to other public annotation software. Our findings suggest that other software incorporating SAM 1 and SAM 2 would benefit from similar gains in annotation speed.</p>
<p>Increasing the annotation efficiency can be especially relevant in fields where human expert knowledge is required. Besides plant phenotyping, one such field would be medicine, where, e.g., radiologists have to label malignant tumor tissue on CT scans (<xref ref-type="bibr" rid="ref49">Zhu et al., 2024</xref>). Also, the example of maize kernels (MED) showed the challenges of qualified decision-making. Although all participants saw the same example masks of valid and invalid maize kernels during a tutorial, the decision on which of the top kernels shown on the left and center pair of maize ears in <xref ref-type="fig" rid="fig5">Figure 5</xref> represent valid kernels was ambiguous. In a scenario where, e.g., the length of the infertile tip of a maize ear should be measured (<xref ref-type="bibr" rid="ref24">Oury et al., 2022</xref>), inconsistent decisions on which kernel to label as valid or fertile could have a crucial impact on the results. However, the agricultural experts that participated in this study were not specifically experts for maize ear phenotyping. Even for professionals in that field, borderline cases and human errors cannot be ruled out for any annotator.</p>
<p>Despite the tutorial covering all functionalities of ARAMSAM, human errors could be observed at the user experiment depicted as outliers in <xref ref-type="fig" rid="fig6">Figures 6</xref>, <xref ref-type="fig" rid="fig7">7</xref>. A few participants appeared to be stuck in certain steps of the experiment, leading them to spend exceptionally long in some annotation tools. Since these outliers indicate a certain complexity of the experiment and do not represent measurement errors, they were included in the statistical analysis. However, the rare occurrence of these outliers shows that only few users encountered these difficulties and most of them were able to learn ARAMSAM quickly.</p>
<p>The integration of the AMG, interactive prompting and mask transfer options for both SAM 1 and SAM 2 as well as polygon drawing as a baseline demonstrates the versatility of ARAMSAM in annotating segmentation datasets. Since the source code of the software will be published along with this paper and since ARAMSAM is completely written in Python, it will be relatively easy to adapt to specific demands. Besides the here conducted annotation experiments ARAMSAM, can be directly used for annotating single images and image sequences for the purpose of training a specific AI model. Such image sequences could include videos, overlapping neighboring images or slices of 3D-data such as CT-Scans or polygon meshes. Also using ARAMSAM directly for measurements of objects in images would be feasible, if intrinsic as well as extrinsic camera parameters and the distance to the object of interest are known.</p>
<p>The development of novel AI-based phenotyping solutions could benefit greatly from accelerated mask annotation based on ARAMSAM. Although <xref ref-type="bibr" rid="ref29">Ravi et al. (2024)</xref> state that mask propagation would suffer from crowded scenes with many object instances, the findings of this study on the MED suggest successful mask propagation in most cases. On average, 94.0% of the masks annotated on the second images originated from mask propagation (<xref ref-type="fig" rid="fig7">Figure 7</xref>). It should be noted that the maize ears are rotated by only 7.1&#x00B0;, leaving a substantial overlap between images to be exploited by SAM 2. Yet, this overlap might be smaller than that of consecutive video frames, for which the SAM 2 mask propagation was designed for (<xref ref-type="bibr" rid="ref29">Ravi et al., 2024</xref>). How far this overlap can be reduced remains an open research question. For both annotation tasks and zero-shot applications, a falsely propagated mask can have a negative impact. However, an overlap of around 80% is, e.g., common in UAV missions for creating digital surface models based on photogrammetry (<xref ref-type="bibr" rid="ref23">Oehme et al., 2022</xref>). This substantial overlap suggests potential for mask propagation with SAM 2 on field image data captured by UAV.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec18">
<label>5</label>
<title>Conclusion</title>
<p>In this study, the potential of both SAM 1 and SAM 2 to accelerate the annotation of segmentation masks as orchestrated by ARAMSAM was evaluated. The annotation time was accelerated by up to 4.6 times (to 2.1&#x202F;s/mask) with SAM 1 on single images and up to 6.1 times (to 1.6&#x202F;s/mask) with SAM 2 on image sequences when compared to polygon drawing, representing remarkable time savings. Moreover, the results on zero-shot performance and from user experiments applying SAM 2 on single images suggest, in accordance with the literature, that SAM 2 represents no improvement on agricultural datasets over SAM 1. In future research on annotation methods in the agricultural domain, finetuning SAM 2 could further accelerate the annotation process.</p>
<p>Furthermore, the importance of hyperparameter optimization of the AMG of both SAM 1 and SAM 2 was demonstrated. The <italic>F<sub>2</sub></italic>-score of predicted masks by SAM 2 when matched to ground-truth masks has been improved by more than 14 times (from 0.05 to 0.74) via grid search for optimal hyperparameters. Moreover, efficient optimization techniques covering larger search spaces such as evolutionary algorithms could be applied in future studies using the AMG.</p>
<p>ARAMSAM, which was developed in this study, is a flexible framework that provides user-friendly access to tools based on SAM 1 and SAM 2. However, the annotation acceleration of SAM 1 and SAM 2 should be further quantified on more diverse and challenging agricultural datasets than those presented in this study. Furthermore, the annotation capabilities of ARAMSAM remain to be compared to other public annotation software in a future study. Nevertheless, built on a Python foundation, ARAMSAM is easily extendable with custom code, allowing researchers to tailor its functionalities to specific needs. Future implementations may include the ability to assign classes to segmentation masks, enriching the software&#x2019;s annotation capabilities. Moreover, ARAMSAM could be integrated with active learning approaches by incorporating pretrained models, which would facilitate the iterative refinement of model performance.</p>
<p>Overall, ARAMSAM, as being published along with this study, is a powerful software solution that integrates the ground-breaking functionalities of both SAM 1 and SAM 2, while also possessing the potential to evolve and make a significant impact on machine vision in agriculture and beyond.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec19">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found at: <ext-link xlink:href="https://github.com/DerOehmer/ARAMSAM/releases/tag/preprint_v0.1" ext-link-type="uri">https://github.com/DerOehmer/ARAMSAM/releases/tag/preprint_v0.1</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="sec20">
<title>Ethics statement</title>
<p>Ethical review and approval was not required for the study on human participants in accordance with the local legislation and institutional requirements. Written informed consent from the participants was not required to participate in this study in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="sec21">
<title>Author contributions</title>
<p>LO: Writing &#x2013; review &#x0026; editing, Writing &#x2013; original draft, Formal analysis, Investigation, Software, Data curation, Validation, Visualization, Methodology, Conceptualization. JB: Methodology, Software, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing, Data curation, Investigation, Conceptualization, Validation. ZW: Writing &#x2013; review &#x0026; editing, Writing &#x2013; original draft. AS: Conceptualization, Resources, Writing &#x2013; review &#x0026; editing, Writing &#x2013; original draft, Project administration, Supervision. JM: Supervision, Conceptualization, Project administration, Writing &#x2013; review &#x0026; editing, Writing &#x2013; original draft, Funding acquisition, Resources.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>A major portion of the content of this manuscript has previously appeared online as a preprint (<xref ref-type="bibr" rid="ref22">Oehme et al., 2025</xref>). The authors would like to thank all participants of the user experiment. Furthermore, the language editing done by Greta Kanzelmeier and Saike Jiang is highly appreciated.</p>
</ack>
<sec sec-type="COI-statement" id="sec22">
<title>Conflict of interest</title>
<p>The author(s) declared that this work was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec23">
<title>Generative AI statement</title>
<p>The author(s) declared that Generative AI was used in the creation of this manuscript. During the preparation of this work the authors used ChatGPT 4o (Open AI, Inc., San Francisco, U.S.) in order to improve written language. After using this tool/service, the authors reviewed and edited the content as needed and take full responsibility for the content of the published article.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec24">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec25"><title>Supplementary material</title><p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2025.1748468/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/frai.2025.1748468/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Image_1.TIF" mimetype="image/TIFF" xmlns:xlink="http://www.w3.org/1999/xlink" id="SM1"><label>Supplementary Figure S1</label><caption><p>Example of maize ear images used for hyperparameter optimization of automatic mask generators (AMG). <bold>(a)</bold> Original RGB image. <bold>(b)</bold> Previously annotated maize kernel instances highlighted by random colors.</p></caption></supplementary-material>
<supplementary-material xlink:href="Image_2.TIF" mimetype="image/TIFF" xmlns:xlink="http://www.w3.org/1999/xlink" id="SM2"><label>Supplementary Figure S2</label><caption><p>Annotation decisions on selected image crop from the leftmost ear in <xref ref-type="fig" rid="fig5">Figure 5</xref>. The colormap on the left side depicts the frequency a pixel has been assigned to a mask relative to the number of annotation rounds (<italic>f<sub>a, px</sub></italic>) Only pixels assigned to a mask more than once are included. The right side shows the original RGB image.</p></caption></supplementary-material>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abbasi</surname><given-names>R.</given-names></name> <name><surname>Martinez</surname><given-names>P.</given-names></name> <name><surname>Ahmad</surname><given-names>R.</given-names></name></person-group> (<year>2022</year>). <article-title>The digitization of the agricultural industry &#x2013; a systematic literature review on agriculture 4.0</article-title>. <source>Smart Agric. Technol.</source> <volume>2</volume>:<fpage>100042</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.atech.2022.100042</pub-id></mixed-citation></ref>
<ref id="ref2"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boysen</surname><given-names>J.</given-names></name> <name><surname>Zender</surname><given-names>L.</given-names></name> <name><surname>Stein</surname><given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Modeling the soil-machine response of secondary tillage: a deep learning approach</article-title>. <source>Smart Agric. Technol.</source> <volume>6</volume>:<fpage>100363</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.atech.2023.100363</pub-id></mixed-citation></ref>
<ref id="ref3"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bradski</surname><given-names>G.</given-names></name></person-group> (<year>2000</year>). <article-title>The OpenCV library</article-title>. <source>Dr. Dobb&#x2019;s J. Softw. Tools</source>.</mixed-citation></ref>
<ref id="ref4"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname><given-names>M.</given-names></name> <name><surname>Lowe</surname><given-names>D. G.</given-names></name></person-group> (<year>2007</year>). <article-title>Automatic panoramic image stitching using invariant features</article-title>. <source>Int. J. Comput. Vis.</source> <volume>74</volume>, <fpage>59</fpage>&#x2013;<lpage>73</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11263-006-0002-3</pub-id></mixed-citation></ref>
<ref id="ref5"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Calonder</surname><given-names>M.</given-names></name> <name><surname>Lepetit</surname><given-names>V.</given-names></name> <name><surname>Strecha</surname><given-names>C.</given-names></name> <name><surname>Fua</surname><given-names>P.</given-names></name></person-group> (<year>2010</year>). &#x201C;<article-title>Brief: binary robust independent elementary features</article-title>&#x201D; in <source>Computer vision&#x2013;ECCV 2010: 11th European conference on computer vision, Heraklion, Crete, Greece, September 5&#x2013;11, 2010, Proceedings, Part IV 11</source> (<publisher-name>Berlin, Heidelberg: Springer</publisher-name>). doi: <pub-id pub-id-type="doi">10.1007/978-3-642-15561-1_56</pub-id></mixed-citation></ref>
<ref id="ref6"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Cardoso</surname><given-names>M. J.</given-names></name> <name><surname>Li</surname><given-names>W.</given-names></name> <name><surname>Brown</surname><given-names>R.</given-names></name> <name><surname>Ma</surname><given-names>N.</given-names></name> <name><surname>Kerfoot</surname><given-names>E.</given-names></name> <name><surname>Wang</surname><given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Monai: an open-source framework for deep learning in healthcare</article-title>, <comment>arXiv preprint arXiv:2211.02701 [Preprint]</comment>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2211.02701</pub-id></mixed-citation></ref>
<ref id="ref7"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Castrej&#x00F3;n</surname><given-names>L.</given-names></name> <name><surname>Kundu</surname><given-names>K.</given-names></name> <name><surname>Urtasun</surname><given-names>R.</given-names></name> <name><surname>Fidler</surname><given-names>S.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Annotating object instances with a polygon-RNN</article-title>&#x201D; in <source>2017 IEEE conference on computer vision and pattern recognition (CVPR)</source>.</mixed-citation></ref>
<ref id="ref8"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dandrifosse</surname><given-names>S.</given-names></name> <name><surname>Ennadifi</surname><given-names>E.</given-names></name> <name><surname>Carlier</surname><given-names>A.</given-names></name> <name><surname>Gosselin</surname><given-names>B.</given-names></name> <name><surname>Dumont</surname><given-names>B.</given-names></name> <name><surname>Mercatoris</surname><given-names>B.</given-names></name></person-group> (<year>2022</year>). <article-title>Deep learning for wheat ear segmentation and ear density measurement: from heading to maturity</article-title>. <source>Comput. Electron. Agric.</source> <volume>199</volume>:<fpage>107161</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2022.107161</pub-id></mixed-citation></ref>
<ref id="ref9"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Daviet</surname><given-names>B.</given-names></name> <name><surname>Fernandez</surname><given-names>R.</given-names></name> <name><surname>Cabrera-Bosquet</surname><given-names>L.</given-names></name> <name><surname>Pradal</surname><given-names>C.</given-names></name> <name><surname>Fournier</surname><given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>PhenoTrack3D: an automatic high-throughput phenotyping pipeline to track maize organs over time</article-title>. <source>Plant Methods</source> <volume>18</volume>:<fpage>130</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13007-022-00961-4</pub-id>, <pub-id pub-id-type="pmid">36482291</pub-id></mixed-citation></ref>
<ref id="ref10"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Dodge</surname><given-names>S.</given-names></name> <name><surname>Karam</surname><given-names>L.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Understanding how image quality affects deep neural networks</article-title>&#x201D; in <source>2016 Eighth international conference on quality of multimedia experience (QoMEX)</source> (<publisher-name>Lisbon, Portugal: IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</mixed-citation></ref>
<ref id="ref11"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname><given-names>A.</given-names></name> <name><surname>Beyer</surname><given-names>L.</given-names></name> <name><surname>Kolesnikov</surname><given-names>A.</given-names></name> <name><surname>Weissenborn</surname><given-names>D.</given-names></name> <name><surname>Zhai</surname><given-names>X.</given-names></name> <name><surname>Unterthiner</surname><given-names>T.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>An image is worth 16x16 words: transformers for image recognition at scale</article-title>, <comment>arXiv preprint arXiv:2010.11929 [Preprint]</comment>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id></mixed-citation></ref>
<ref id="ref12"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Farooq</surname><given-names>M. A.</given-names></name> <name><surname>Gao</surname><given-names>S.</given-names></name> <name><surname>Hassan</surname><given-names>M. A.</given-names></name> <name><surname>Huang</surname><given-names>Z.</given-names></name> <name><surname>Rasheed</surname><given-names>A.</given-names></name> <name><surname>Hearne</surname><given-names>S.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Artificial intelligence in plant breeding</article-title>. <source>Trends Genet.</source> <volume>40</volume>, <fpage>891</fpage>&#x2013;<lpage>908</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tig.2024.07.001</pub-id>, <pub-id pub-id-type="pmid">39117482</pub-id></mixed-citation></ref>
<ref id="ref13"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hunter</surname><given-names>J. D.</given-names></name></person-group> (<year>2007</year>). <article-title>Matplotlib: a 2D graphics environment</article-title>. <source>Comput. Sci. Eng.</source> <volume>9</volume>, <fpage>90</fpage>&#x2013;<lpage>95</lpage>. doi: <pub-id pub-id-type="doi">10.1109/MCSE.2007.55</pub-id></mixed-citation></ref>
<ref id="ref14"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kassambara</surname><given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>rstatix: pipe-friendly framework for basic statistical tests</article-title>. <source>CRAN: Contrib. Packages</source>. doi: <pub-id pub-id-type="doi">10.32614/cran.package.rstatix</pub-id></mixed-citation></ref>
<ref id="ref15"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kirchgessner</surname><given-names>N.</given-names></name> <name><surname>Hodel</surname><given-names>M.</given-names></name> <name><surname>Studer</surname><given-names>B.</given-names></name> <name><surname>Patocchi</surname><given-names>A.</given-names></name> <name><surname>Broggini</surname><given-names>G. A. L.</given-names></name></person-group> (<year>2024</year>). <article-title>FruitPhenoBox &#x2013; a device for rapid and automated fruit phenotyping of small sample sizes</article-title>. <source>Plant Methods</source> <volume>20</volume>:<fpage>74</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13007-024-01206-2</pub-id>, <pub-id pub-id-type="pmid">38783345</pub-id></mixed-citation></ref>
<ref id="ref16"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Kirillov</surname><given-names>A.</given-names></name> <name><surname>Mintun</surname><given-names>E.</given-names></name> <name><surname>Ravi</surname><given-names>N.</given-names></name> <name><surname>Mao</surname><given-names>H.</given-names></name> <name><surname>Rolland</surname><given-names>C.</given-names></name> <name><surname>Gustafson</surname><given-names>L.</given-names></name></person-group> (<year>2023</year>). &#x201C;<article-title>Segment anything</article-title>&#x201D; in <source>Proceedings of the IEEE/CVF international conference on computer vision</source>.</mixed-citation></ref>
<ref id="ref17"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Manzini</surname><given-names>T.</given-names></name> <name><surname>Perali</surname><given-names>P.</given-names></name> <name><surname>Karnik</surname><given-names>R.</given-names></name> <name><surname>Godbole</surname><given-names>M.</given-names></name> <name><surname>Abdullah</surname><given-names>H.</given-names></name> <name><surname>Murphy</surname><given-names>R.</given-names></name></person-group> (<year>2024</year>). <article-title>Non-uniform spatial alignment errors in sUAS imagery from wide-area disasters</article-title>, <comment>arXiv preprint arXiv:2405.06593 [Preprint]</comment>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2405.06593</pub-id></mixed-citation></ref>
<ref id="ref18"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maraveas</surname><given-names>C.</given-names></name></person-group> (<year>2024</year>). <article-title>Image analysis artificial intelligence technologies for plant phenotyping: current state of the art</article-title>. <source>AgriEngineering</source> <volume>6</volume>, <fpage>3375</fpage>&#x2013;<lpage>3407</lpage>. doi: <pub-id pub-id-type="doi">10.3390/agriengineering6030193</pub-id></mixed-citation></ref>
<ref id="ref19"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>McKinney</surname><given-names>W.</given-names></name></person-group> (<year>2010</year>). &#x201C;<article-title>Data structures for statistical computing in Python</article-title>&#x201D; in <source>9th Python in science conference</source>.</mixed-citation></ref>
<ref id="ref20"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Minervini</surname><given-names>M.</given-names></name> <name><surname>Fischbach</surname><given-names>A.</given-names></name> <name><surname>Scharr</surname><given-names>H.</given-names></name> <name><surname>Tsaftaris</surname><given-names>S. A.</given-names></name></person-group> (<year>2016</year>). <article-title>Finely-grained annotated datasets for image-based plant phenotyping</article-title>. <source>Pattern Recogn. Lett.</source> <volume>81</volume>, <fpage>80</fpage>&#x2013;<lpage>89</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.patrec.2015.10.013</pub-id></mixed-citation></ref>
<ref id="ref21"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moreira</surname><given-names>G.</given-names></name> <name><surname>dos Santos</surname><given-names>F. N.</given-names></name> <name><surname>Cunha</surname><given-names>M.</given-names></name></person-group> (<year>2025</year>). <article-title>Grapevine inflorescence segmentation and flower estimation based on computer vision techniques for early yield assessment</article-title>. <source>Smart Agric. Technol.</source> <volume>10</volume>:<fpage>100690</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.atech.2024.100690</pub-id></mixed-citation></ref>
<ref id="ref22"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oehme</surname><given-names>L. H.</given-names></name> <name><surname>Boysen</surname><given-names>J.</given-names></name> <name><surname>Wu</surname><given-names>Z.</given-names></name> <name><surname>Stein</surname><given-names>A.</given-names></name> <name><surname>M&#x00FC;ller</surname><given-names>J.</given-names></name></person-group> (<year>2025</year>). <article-title>Orchestrating segment anything models to accelerate segmentation annotation on agricultural image datasets</article-title>. <source>Res. Sq.</source> doi: <pub-id pub-id-type="doi">10.21203/rs.3.rs-7606794/v1</pub-id></mixed-citation></ref>
<ref id="ref23"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oehme</surname><given-names>L. H.</given-names></name> <name><surname>Reineke</surname><given-names>A.-J.</given-names></name> <name><surname>Wei&#x00DF;</surname><given-names>T. M.</given-names></name> <name><surname>W&#x00FC;rschum</surname><given-names>T.</given-names></name> <name><surname>He</surname><given-names>X.</given-names></name> <name><surname>M&#x00FC;ller</surname><given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Remote sensing of maize plant height at different growth stages using UAV-based digital surface models (DSM)</article-title>. <source>Agronomy</source> <volume>12</volume>:<fpage>958</fpage>. doi: <pub-id pub-id-type="doi">10.3390/agronomy12040958</pub-id></mixed-citation></ref>
<ref id="ref24"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oury</surname><given-names>V.</given-names></name> <name><surname>Leroux</surname><given-names>T.</given-names></name> <name><surname>Turc</surname><given-names>O.</given-names></name> <name><surname>Chapuis</surname><given-names>R.</given-names></name> <name><surname>Palaffre</surname><given-names>C.</given-names></name> <name><surname>Tardieu</surname><given-names>F.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Earbox, an open tool for high-throughput measurement of the spatial organization of maize ears and inference of novel traits</article-title>. <source>Plant Methods</source> <volume>18</volume>:<fpage>96</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13007-022-00925-8</pub-id>, <pub-id pub-id-type="pmid">35902871</pub-id></mixed-citation></ref>
<ref id="ref25"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pal</surname><given-names>J. B.</given-names></name> <name><surname>Bhattacharyea</surname><given-names>A.</given-names></name> <name><surname>Banerjee</surname><given-names>D.</given-names></name> <name><surname>Maharaj</surname><given-names>B. T.</given-names></name></person-group> (<year>2024</year>). <article-title>Advancing instance segmentation and WBC classification in peripheral blood smear through domain adaptation: a study on PBC and the novel RV-PBS datasets</article-title>. <source>Expert Syst. Appl.</source> <volume>249</volume>:<fpage>123660</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.eswa.2024.123660</pub-id></mixed-citation></ref>
<ref id="ref26"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paszke</surname><given-names>A.</given-names></name> <name><surname>Gross</surname><given-names>S.</given-names></name> <name><surname>Massa</surname><given-names>F.</given-names></name> <name><surname>Lerer</surname><given-names>A.</given-names></name> <name><surname>Bradbury</surname><given-names>J.</given-names></name> <name><surname>Chanan</surname><given-names>G.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Pytorch: an imperative style, high-performance deep learning library</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>32</volume>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1912.01703</pub-id></mixed-citation></ref>
<ref id="ref27"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paton</surname><given-names>N. W.</given-names></name> <name><surname>Chen</surname><given-names>J.</given-names></name> <name><surname>Wu</surname><given-names>Z.</given-names></name></person-group> (<year>2024</year>). <article-title>Dataset discovery and exploration: a survey</article-title>. <source>ACM Comput. Surv.</source> <volume>56</volume>, <fpage>1</fpage>&#x2013;<lpage>37</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3626521</pub-id></mixed-citation></ref>
<ref id="ref28"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Rafaeli</surname><given-names>O.</given-names></name> <name><surname>Svoray</surname><given-names>T.</given-names></name> <name><surname>Blushtein-Livnon</surname><given-names>R.</given-names></name> <name><surname>Nahlieli</surname><given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>Prompt-based segmentation at multiple resolutions and lighting conditions using segment anything model 2</article-title>, <comment>arXiv preprint arXiv:2408.06970 [Preprint]</comment>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2408.06970</pub-id></mixed-citation></ref>
<ref id="ref29"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Ravi</surname><given-names>N.</given-names></name> <name><surname>Gabeur</surname><given-names>V.</given-names></name> <name><surname>Hu</surname><given-names>Y.-T.</given-names></name> <name><surname>Hu</surname><given-names>R.</given-names></name> <name><surname>Ryali</surname><given-names>C.</given-names></name> <name><surname>Ma</surname><given-names>T.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Sam 2: segment anything in images and videos</article-title>, <comment>arXiv preprint arXiv:2408.00714 [Preprint]</comment>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2408.00714</pub-id></mixed-citation></ref>
<ref id="ref30"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rejeb</surname><given-names>A.</given-names></name> <name><surname>Abdollahi</surname><given-names>A.</given-names></name> <name><surname>Rejeb</surname><given-names>K.</given-names></name> <name><surname>Treiblmaier</surname><given-names>H.</given-names></name></person-group> (<year>2022</year>). <article-title>Drones in agriculture: a review and bibliometric analysis</article-title>. <source>Comput. Electron. Agric.</source> <volume>198</volume>:<fpage>107017</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2022.107017</pub-id></mixed-citation></ref>
<ref id="ref47"><mixed-citation publication-type="book"><person-group person-group-type="author"><collab id="coll1">Riverbank Computing</collab></person-group>. (<year>2025</year>). <source>PyQt6</source>. <edition>6.7</edition> Edn: <publisher-name>Riverbank Computing Limited</publisher-name>.</mixed-citation></ref>
<ref id="ref31"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Rosten</surname><given-names>E.</given-names></name> <name><surname>Drummond</surname><given-names>T.</given-names></name></person-group> (<year>2006</year>). &#x201C;<article-title>Machine learning for high-speed corner detection</article-title>&#x201D; in <source>Computer vision&#x2013;ECCV 2006: 9th European conference on computer vision, Graz, Austria, may 7&#x2013;13, 2006. Proceedings, part I 9</source> (<publisher-name>Berlin, Heidelberg: Springer</publisher-name>). doi: <pub-id pub-id-type="doi">10.1007/11744023_34</pub-id></mixed-citation></ref>
<ref id="ref32"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Rublee</surname><given-names>E.</given-names></name> <name><surname>Rabaud</surname><given-names>V.</given-names></name> <name><surname>Konolige</surname><given-names>K.</given-names></name> <name><surname>Bradski</surname><given-names>G.</given-names></name></person-group> (<year>2011</year>). &#x201C;<article-title>ORB: an efficient alternative to SIFT or SURF</article-title>&#x201D; in <source>2011 International conference on computer vision</source> (<publisher-name>IEEE</publisher-name>).</mixed-citation></ref>
<ref id="ref33"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Schonberger</surname><given-names>J. L.</given-names></name> <name><surname>Frahm</surname><given-names>J.-M.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Structure-from-motion revisited</article-title>&#x201D; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>4104</fpage>&#x2013;<lpage>4113</lpage>.</mixed-citation></ref>
<ref id="ref34"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Sengupta</surname><given-names>S.</given-names></name> <name><surname>Chakrabarty</surname><given-names>S.</given-names></name> <name><surname>Soni</surname><given-names>R.</given-names></name></person-group> (<year>2025</year>). &#x201C;<article-title>Is SAM 2 better than SAM in medical image segmentation?</article-title>&#x201D; in <source>Med. Imaging 2025: image process 13406</source>, <fpage>666</fpage>&#x2013;<lpage>672</lpage>.</mixed-citation></ref>
<ref id="ref35"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Shao</surname><given-names>H.-C.</given-names></name> <name><surname>Lin</surname><given-names>Y.-H.</given-names></name> <name><surname>Lin</surname><given-names>C.-W.</given-names></name></person-group> (<year>2024</year>). &#x201C;<article-title>A fine-grained attribute pre-labeling method based on label dependency and feature similarity dynamics</article-title>&#x201D; in <source>ICASSP 2024&#x2013;2024 IEEE international conference on acoustics, speech and signal processing (ICASSP)</source> (<publisher-name>IEEE</publisher-name>).</mixed-citation></ref>
<ref id="ref36"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sheikh</surname><given-names>M.</given-names></name> <name><surname>Iqra</surname><given-names>F.</given-names></name> <name><surname>Ambreen</surname><given-names>H.</given-names></name> <name><surname>Pravin</surname><given-names>K. A.</given-names></name> <name><surname>Ikra</surname><given-names>M.</given-names></name> <name><surname>Chung</surname><given-names>Y. S.</given-names></name></person-group> (<year>2024</year>). <article-title>Integrating artificial intelligence and high-throughput phenotyping for crop improvement</article-title>. <source>J. Integr. Agric.</source> <volume>23</volume>, <fpage>1787</fpage>&#x2013;<lpage>1802</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jia.2023.10.019</pub-id></mixed-citation></ref>
<ref id="ref37"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Song</surname><given-names>X.</given-names></name> <name><surname>Yan</surname><given-names>L.</given-names></name> <name><surname>Liu</surname><given-names>S.</given-names></name> <name><surname>Gao</surname><given-names>T.</given-names></name> <name><surname>Han</surname><given-names>L.</given-names></name> <name><surname>Jiang</surname><given-names>X.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Agricultural image processing: challenges, advances, and future trends</article-title>. <source>Appl. Sci.</source> <volume>15</volume>:<fpage>9206</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app15169206</pub-id></mixed-citation></ref>
<ref id="ref38"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Sudre</surname><given-names>C. H.</given-names></name> <name><surname>Li</surname><given-names>W.</given-names></name> <name><surname>Vercauteren</surname><given-names>T.</given-names></name> <name><surname>Ourselin</surname><given-names>S.</given-names></name> <name><surname>Jorge Cardoso</surname><given-names>M.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Generalised dice overlap as a deep learning loss function for highly unbalanced segmentations</article-title>&#x201D; in <source>Deep learning in medical image analysis and multimodal learning for clinical decision support: Third international workshop, DLMIA 2017, and 7th international workshop, ML-CDS 2017, held in conjunction with MICCAI 2017, Qu&#x00E9;bec City, QC, Canada, September 14, proceedings 3</source> (<publisher-name>Cham: Springer</publisher-name>). doi: <pub-id pub-id-type="doi">10.1007/978-3-319-67558-9_28</pub-id></mixed-citation></ref>
<ref id="ref39"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Team</surname><given-names>R.C.</given-names></name></person-group> (<year>2025</year>). <source>R: a language and environment for statistical computing</source>.</mixed-citation></ref>
<ref id="ref40"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Van Rossum</surname><given-names>G.</given-names></name> <name><surname>Drake</surname><given-names>F. L.</given-names></name></person-group> (<year>2009</year>). <source>Introduction to python 3: python documentation manual part 1</source>: <publisher-name>CreateSpace</publisher-name>.</mixed-citation></ref>
<ref id="ref41"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vardi</surname><given-names>Y.</given-names></name> <name><surname>Zhang</surname><given-names>C.-H.</given-names></name></person-group> (<year>2000</year>). <article-title>The multivariate L 1-median and associated data depth</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>97</volume>, <fpage>1423</fpage>&#x2013;<lpage>1426</lpage>. doi: <pub-id pub-id-type="doi">10.1073/pnas.97.4.1423</pub-id>, <pub-id pub-id-type="pmid">10677477</pub-id></mixed-citation></ref>
<ref id="ref42"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Visakh</surname><given-names>R. L.</given-names></name> <name><surname>Anand</surname><given-names>S.</given-names></name> <name><surname>Reddy</surname><given-names>S. B.</given-names></name> <name><surname>Jha</surname><given-names>U. C.</given-names></name> <name><surname>Sah</surname><given-names>R. P.</given-names></name> <name><surname>Beena</surname><given-names>R.</given-names></name></person-group> (<year>2024</year>). <article-title>Precision phenotyping in crop science: from plant traits to gene discovery for climate-smart agriculture</article-title>. <source>Plant Breed.</source> doi: <pub-id pub-id-type="doi">10.1111/pbr.13228</pub-id></mixed-citation></ref>
<ref id="ref43"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Wada</surname><given-names>K.</given-names></name></person-group> (<year>2025</year>). <source>Labelme: image polygonal annotation with Python</source>.</mixed-citation></ref>
<ref id="ref44"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Waskom</surname><given-names>M.</given-names></name> <name><surname>Botvinnik</surname><given-names>O.</given-names></name> <name><surname>O'Kane</surname><given-names>D.</given-names></name> <name><surname>Hobson</surname><given-names>P.</given-names></name> <name><surname>Lukauskas</surname><given-names>S.</given-names></name> <name><surname>Gemperline</surname><given-names>D. C.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Seaborn: statistical data visualization</article-title>. <source>J. Open Source Softw.</source> <volume>6</volume>. doi: <pub-id pub-id-type="doi">10.21105/joss.03021</pub-id></mixed-citation></ref>
<ref id="ref45"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wickham</surname><given-names>H.</given-names></name> <name><surname>Averick</surname><given-names>M.</given-names></name> <name><surname>Bryan</surname><given-names>J.</given-names></name> <name><surname>Chang</surname><given-names>W.</given-names></name> <name><surname>McGowan</surname><given-names>L. D. A.</given-names></name> <name><surname>Fran&#x00E7;ois</surname><given-names>R.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Welcome to the Tidyverse</article-title>. <source>J. Open Source Softw.</source> <volume>4</volume>. doi: <pub-id pub-id-type="doi">10.21105/joss.01686</pub-id></mixed-citation></ref>
<ref id="ref46"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Williams</surname><given-names>D.</given-names></name> <name><surname>Macfarlane</surname><given-names>F.</given-names></name> <name><surname>Britten</surname><given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>Leaf only SAM: a segment anything pipeline for zero-shot automated leaf segmentation</article-title>. <source>Smart Agric. Technol.</source> <volume>8</volume>:<fpage>100515</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.atech.2024.100515</pub-id></mixed-citation></ref>
<ref id="ref48"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>W.</given-names></name> <name><surname>Dang</surname><given-names>L. M.</given-names></name> <name><surname>Nguyen</surname><given-names>L. Q.</given-names></name> <name><surname>Alam</surname><given-names>N.</given-names></name> <name><surname>Bui</surname><given-names>N. D.</given-names></name> <name><surname>Park</surname><given-names>H. Y.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Adapting the segment anything model for plant recognition and automated phenotypic parameter measurement</article-title>. <source>Horticulturae</source> <volume>10</volume>:<fpage>398</fpage>. doi: <pub-id pub-id-type="doi">10.3390/horticulturae10040398</pub-id></mixed-citation></ref>
<ref id="ref49"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Zhu</surname><given-names>J.</given-names></name> <name><surname>Hamdi</surname><given-names>A.</given-names></name> <name><surname>Qi</surname><given-names>Y.</given-names></name> <name><surname>Jin</surname><given-names>Y.</given-names></name> <name><surname>Wu</surname><given-names>J.</given-names></name></person-group> (<year>2024</year>). <article-title>Medical SAM 2: segment medical images as video via segment anything model 2</article-title>. <comment>arXiv preprint arXiv:2408.00874</comment>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2408.00874</pub-id></mixed-citation></ref>
<ref id="ref50"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname><given-names>K. H.</given-names></name> <name><surname>Warfield</surname><given-names>S. K.</given-names></name> <name><surname>Bharatha</surname><given-names>A.</given-names></name> <name><surname>Tempany</surname><given-names>C. M. C.</given-names></name> <name><surname>Kaus</surname><given-names>M. R.</given-names></name> <name><surname>Haker</surname><given-names>S. J.</given-names></name> <etal/></person-group>. (<year>2004</year>). <article-title>Statistical validation of image segmentation quality based on a spatial overlap index1: scientific reports</article-title>. <source>Acad. Radiol.</source> <volume>11</volume>, <fpage>178</fpage>&#x2013;<lpage>189</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S1076-6332(03)00671-8</pub-id></mixed-citation></ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0001"><p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3152000/overview">Sathishkumar Samiappan</ext-link>, The University of Tennessee, Knoxville, United States</p></fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0002"><p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/159803/overview">Chrysanthos Maraveas</ext-link>, Agricultural University of Athens, Greece</p><p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3306338/overview">Jesus Franco-Robles</ext-link>, UMR7252 XLIM, France</p></fn>
</fn-group>
<fn-group>
<fn fn-type="abbr" id="abbrev1"><label>Abbreviations:</label><p>AI, artificial intelligence; AMG, automatic mask generator; ARAMSAM, agricultural rapid annotation module based on segment anything models; BRIEF, binary robust independent elementary features; FAST, features from the accelerated segment test; GDS, generalized dice score; Hiera, hierarchical vision transformers; IoU, intersection over union; mIoU, mean intersection over union; MED, maize ear dataset; MUD, maize field UAV dataset; ORB, oriented FAST and rotated BRIEF; PPDLS, plant phenotyping datasets leaf segmentation; SAM 1, SAM 2, segment anything models; SOD, soil surface dataset; UAV, unmanned aerial vehicle; ViT, vision transformer.</p></fn>
</fn-group>
</back>
</article>