<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<article dtd-version="1.1" xml:lang="en" xmlns="http://jats.nlm.nih.gov" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/">
    <front>
        <journal-meta>
            <journal-id>OENO One</journal-id>
            <issn>2494-1271</issn>
        </journal-meta>
        <article-meta>
            <title-group>
                <article-title xml:lang="en">Grapevine-Seg: A grapevine segmentation method based on an improved YOLACT</article-title>
            </title-group>
            <contrib-group>
                <contrib contrib-type="dc:creator">
                    <name>
                        <surname>Bu</surname>
                        <given-names>Lingxin</given-names>
                    </name>
                    <email>2021023@nmu.edu.cn</email>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>Su</surname>
                        <given-names>Jie</given-names>
                    </name>
                    <email>jiesu9493@163.com</email>
                    <xref ref-type="aff" rid="aff2">
                        <sup>2</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>Zhang</surname>
                        <given-names>Qiangqiang</given-names>
                    </name>
                    <email>18435893228@163.com</email>
                    <xref ref-type="aff" rid="aff3">
                        <sup>3</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>Kou</surname>
                        <given-names>Qianwen</given-names>
                    </name>
                    <email>kouqianwen2023@126.com</email>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>Chen</surname>
                        <given-names>Yun</given-names>
                    </name>
                    <email>935762276@qq.com</email>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>Wang</surname>
                        <given-names>Jipeng</given-names>
                    </name>
                    <email>2357962221@qq.com</email>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>Tang</surname>
                        <given-names>Xingrun</given-names>
                    </name>
                    <email>2082620856@qq.com</email>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>Li</surname>
                        <given-names>Xingjia</given-names>
                    </name>
                    <email>18152382158@163.com</email>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>Wu</surname>
                        <given-names>Xingtao</given-names>
                    </name>
                    <email>18693539213@163.com</email>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>Zhang</surname>
                        <given-names>Teng</given-names>
                    </name>
                    <email>zhangteng1893@163.com</email>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>and</surname>
                        <given-names/>
                    </name>
                </contrib>
                <contrib contrib-type="dc:contributor">
                    <name>
                        <surname>Zhao</surname>
                        <given-names>Li</given-names>
                    </name>
                    <email>linzhaoli@126.com</email>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
            </contrib-group>
            <aff id="aff1">
                <sup>1</sup>College of Mechatronic Engineering, North Minzu University, Yinchuan, Ningxia, 750021, China</aff>
            <aff id="aff2">
                <sup>2</sup>Ningxia Hongyuan Great Wall Machine Tool Co., Ltd., Yinchuan, Ningxia, 750021, China</aff>
            <aff id="aff3">
                <sup>3</sup>School of Mechanical Engineering, Ningxia University, Yinchuan, Ningxia, 750021, China</aff>
            <pub-date date-type="created">
                <day>4</day>
                <month>1</month>
                <year>2026</year>
            </pub-date>
            <permissions>
                <copyright-statement>Copyright © 1970 Lingxin Bu, Jie Su, Qiangqiang Zhang, Qianwen Kou, Yun Chen, Jipeng Wang, Xingrun Tang, Xingjia Li, Xingtao Wu, Teng Zhang, null and, Li Zhao</copyright-statement>
                <copyright-year>1970</copyright-year>
                <copyright-holder>Lingxin Bu, Jie Su, Qiangqiang Zhang, Qianwen Kou, Yun Chen, Jipeng Wang, Xingrun Tang, Xingjia Li, Xingtao Wu, Teng Zhang, null and, Li Zhao</copyright-holder>
                <license>
                    <license-p/>
                </license>
            </permissions>
            <abstract xml:lang="en">
                <p>Real-time segmentation of grapevine cordons and shoots remains a critical bottleneck for automated viticultural management, particularly in resource-constrained field environments where computational efficiency is paramount. This study established a lightweight segmentation model based on an improved YOLACT (You Only Look At Coefficient) framework, optimised for real-time grapevine segmentation on embedded systems. In place of the original ResNet backbone is Ghost-Net, which uses Ghost modules to generate redundant feature maps and produce richer feature representations, thus reducing the computational load of convolutions and improving small-object detection and segmentation. Concurrently, an EMAttention (Emefficient multi-scale attention) module is embedded at the skip connections between the Feature Pyramid Network (FPN) and the Mask Head. This module concatenates the FPN features of multiple scales with Mask Head features, applies spatial convolution to generate spatial attention maps that bolster object-region features, and adaptively fuses multi-scale features with learned weights, thus improving segmentation performance across differently sized objects. The modifications, focusing on lightweight design and inference speed, yield a model that balances accuracy and efficiency more suitably for embedded viticultural systems than existing benchmarks. Under identical experimental conditions, the improved model outperformed mainstream segmentation frameworks. The average detection accuracies for grapevine and lateral shoot test samples reached 69.46 % and 67.66 %, respectively. These results demonstrate the potential of this approach for enabling real-time, field-deployable grapevine segmentation in precision viticulture applications. The annotated dataset (Grapevine-Seg) is publicly available at https://zenodo.org/records/18218165.</p>
            </abstract>
            <kwd-group>
                <kwd>deep</kwd>
                <kwd>learning</kwd>
                <kwd>convolutional</kwd>
                <kwd>network</kwd>
                <kwd>vineyard</kwd>
                <kwd>precision</kwd>
                <kwd>viticulture</kwd>
                <kwd>grapevine</kwd>
                <kwd>pruning</kwd>
                <kwd>deep learning</kwd>
                <kwd>convolutional network</kwd>
                <kwd>vineyard</kwd>
                <kwd>precision viticulture</kwd>
                <kwd>grapevine pruning</kwd>
            </kwd-group>
        </article-meta>
    </front>
    <body>
        <sec id="h0-introduction">
            <title>Introduction</title>
            <p>Grapevines, as economically significant crops, require precise management to optimise yield and improve quality. Grapevine segmentation serves as a core technology enabling precision operations and has been a research focus within agricultural intelligence for several decades. Early pioneering work on vision-guided grapevine pruning can be traced back to Mercurio <italic>et al.</italic> (1989), who developed a block-type robotic pruner with machine vision capabilities. Subsequently, McFarlane <italic>et al.</italic> (1997) explored image analysis techniques for pruning long-wood grape vines, and Gao and Lu (2006) further investigated image processing methods for autonomous grapevine pruning. These early studies highlighted the inherent complexity of extracting meaningful features from vine images and laid the foundation for subsequent advances. With the emergence of deep learning, this field has attracted renewed attention and has recently become an active research area (Botterill <italic>et al.</italic>, 2017). Traditional viticulture relies heavily on manual labour for essential tasks like pruning and flower thinning, which is not only inefficient and costly but also unsustainable given large-scale production demands (Bochtis <italic>et al.</italic>, 2014). Thus, a growing need exists to advance automation and intelligence in viticultural management (Bu <italic>et al.</italic>, 2025; Íñiguez <italic>et al.</italic>, 2025).</p>
            <p>In viticultural management research, Karkee <italic>et al.</italic> (2023) applied robotic precision pruning to planar tree canopies in orchards and vineyards. Their approach optimises yield through load management techniques such as pruning and thinning, thereby increasing productivity and resource utilisation. Furthermore, Guadagna <italic>et al.</italic> (2023) used the Faster Region-based Convolutional Neural Network (Faster R-CNN) model to detect visible inter-mediate complex buds with a detection rate of 0.97 and the most prevalent coplanar simple buds at 74 %. Mask Region-based Convolutional Neural Network (Mask R-CNN) experiments indicated optimal node segmentation and a shoot-thinned recall rate of 0.85, exceeding the control group. Similarly, Majeed <italic>et al.</italic> (2019) developed a Faster R-CNN–based method to detect visible stem segments using transfer learning on pretrained networks. The Residual Network 18 (ResNet-18)-based model performed the best, achieving an F1 score of 0.55 and a mean average precision of 45.1 %. Majeed <italic>et al.</italic> (2020a) used deep learning networks to determine the grapevine’s main stem contour from colour camera imagery, applying SegNet and FCN (Fully Convolutional Networks) segmentation techniques. The model accurately traced the main stem trajectory even when concealed by foliage. In a different study, Majeed <italic>et al.</italic> (2020b) also detected visible segments of the trunk and main stem using Faster R-CNN with a ResNet-18 backbone, applied non-maximum suppression to refine detections, and fitted a 6th-degree polynomial to centroids of detected segments to estimate the main stem trajectory. They found that for vines two to four weeks post-budburst, trajectory estimation yielded correlation coefficients of 0.993, 0.991, and 0.987. Moreover, Marset <italic>et al.</italic> (2021) implemented a fully convolutional network based on MobileNet for bud segmentation. They performed pixel-level classification, followed by post-processing, to establish bud correspondences and centroid localisation. The best FCN-MN (Fully Convolutional Networks MobileNet) model achieved an F1 score of 88.6 %, indicating high segmentation accuracy. Similarly, Moreno and Andújar (2023) reviewed the application of proximal sensing technologies for geometric characterisation of grapevines, concluding that while LiDAR offers high precision but at an elevated cost, ultrasonic sensors are low-cost and low-resolution, and that depth cameras achieve a balance between cost and accuracy. Furthermore, Casado-García <italic>et al.</italic> (2022) compared multiple depth architectures and employed three semi-supervised techniques, including pseudo-labelling, leveraging unlabelled data. The semi-supervised approach increased the average accuracy by 5.62 %–6.01 %. Gentilhomme <italic>et al.</italic> (2023) used the ViNet deep learning approach, employing a stacked hourglass network to detect nodes, classify branch types, and infer spatial relationships. A shortest-path weighted graph algorithm was then employed to optimally extract connections between nodes, yielding a node precision of 95 % and a recall of 90 %. Dong <italic>et al.</italic> (2016) adopted a Mask R-CNN segmentation method to compile a dataset of key grapevine structures and conducted a comparative analysis. Their Mask R-CNN model, featuring a Residual Network 101 (ResNet-101) + feature pyramid network (FPN) backbone, achieved precision, recall, and mean average precision values of 85.04 %, 82.03 %, and 85.40 %, respectively, significantly outperforming comparative models. Recent studies have demonstrated that combining automatic annotation pipelines with transformer-based segmentation frameworks can significantly improve structural delineation and growth analysis in field crops (Rana <italic>et al.</italic>, 2024). Such integrated methods using YOLO-based detectors and SAM (Segment Anything Model) architectures have proven effective in capturing fine structural boundaries across complex agricultural scenes, and similar workflows can be adapted for grapevine canopy and cordon segmentation. More recently, Fernandes <italic>et al.</italic> (2025) demonstrated the potential of merging 2D segmentation with 3D point clouds for pruning point generation, representing a promising direction for integrating multi-modal data in viticultural applications.</p>
            <p>Although existing models have made progress in accuracy and environmental robustness, the dynamic nature of grapevine growth—such as structural variations across phenological stages—and complex environmental interference in field settings (
                <italic>e.g.</italic>, weeds and uneven lighting) continue to limit the practical application of segmentation techniques. To address these challenges, this study introduces an improved YOLACT-based framework by innovatively integrating GhostNet and EMAttention mechanisms. In contrast to YOLACT++, which primarily focuses on backbone refinement, or recent approaches such as ViNet that employ more complex architectures, our lightweight solution leverages Ghost modules to generate redundant feature maps, thereby enhancing small-object feature representation while reducing computational cost. Simultaneously, the EMAttention module enables adaptive fusion of multi-scale features, improving segmentation accuracy across varying growth stages and complex field environments without compromising real-time performance. Further investigation into efficient segmentation algorithms remains essential to advance intelligent viticultural systems.</p>
            <p>This work is practical rather than conceptual. We present an incremental yet effective improvement to the YOLACT architecture, specifically optimised for the challenge of real-time grapevine segmentation in resource-constrained field environments. Specifically, the scientific objectives of this study are: (1) to develop a lightweight segmentation model capable of real-time grapevine cordon and shoot detection suitable for deployment on resource-constrained embedded systems; (2) to evaluate whether the integration of GhostNet backbone and EMAttention mechanism can effectively improve segmentation accuracy for small and variable-sized grapevine structures under complex field conditions; and (3) to establish a publicly available annotated dataset (Grapevine-Seg) that can serve as a benchmark for future research in automated viticultural management.</p>
        </sec>
        <sec id="h1-materials-and-methods">
            <title>Materials and methods</title>
            <sec id="h0-1--image-acquisition-and-preprocessing">
                <title>1. Image acquisition and preprocessing</title>
                <p>Currently, publicly available datasets for grapevine instance segmentation are limited. Gentilhomme <italic>et al.</italic> (2023) collected 1,513 images of grape plants using smartphones or digital cameras, annotating structural elements such as the trunk, cane, shoots, and nodes for structural extraction tasks. In the Helan Mountains west slope region of Ningxia, vineyards primarily leverage a “sloping trunk horizontal cordon” training system, which aids in facilitating vine burial and emergence while mitigating frost injury and desiccation, as shown in Figure 1.</p>
                <p>
                    <fig>
                        <caption>
                            <title>Figure 1. Sample image from the grapevine dataset comprising: (a) the original RGB image of the grapevine plant and (b) the annotated mask with structural labels such as shoots and cordon.</title>
                        </caption>
                        <graphic xlink:href="media/image1.png"/>
                    </fig>
                </p>
                <p>Images were captured in the vineyard of Ningxia Mutong Winery Co., Ltd., at the coordinates 38.61° N 106.13° E. The grape cultivar imaged was Cabernet-Sauvignon, cultivated for 7–8 years. Image capture was conducted using a Huawei Nova 11 smartphone. Each image measured 1,920 × 1,080 pixels, and data collection spanned from 09:00 to 17:00 on October 28, 2023. Then, the original images were annotated using LabelMe. Structural components, including the cordon and shoots, were delineated to produce JSON annotation files, which were then converted into segmentation mask images. The final dataset comprised 2,091 images, which were divided into a training set of 1,672 images and a validation set of 419 images. The dataset is publicly available on 
                    <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/18218165">https://zenodo.org/records/18218165</ext-link> for academic research and can thus be used for benchmarking purposes.</p>
            </sec>
            <sec id="h2-2--improved-yolact-model">
                <title>2. Improved YOLACT model</title>
                <p>YOLACT is the first single-stage algorithm that enables real-time instance segmentation by predicting prototype masks and mask coefficients, which are linearly combined to synthesise the final segmentation masks (Bolya <italic>et al.</italic>, 2019). Its key components are the backbone network, feature pyramid network (FPN), Detection Head, and Mask Head. However, its architecture presents several limitations that hinder its effectiveness in practical applications like automated viticulture: (1) The commonly used ResNet-50 backbone suffers from low efficiency, with 25.6 M parameters and 3.8 G FLOPs, creating significant computational redundancy and hindering deployment on edge devices. (2) The feature pyramid network (FPN) primarily performs multi-scale fusion but lacks sophisticated attention mechanisms to model channel-wise semantics and spatial instance specificity, limiting its feature representation power. (3) The simplified prototype mask generation, a design choice for speed, results in insufficient capture of fine-grained features, leading to a suboptimal speed-accuracy trade-off. To address the identified limitations, two key enhancements are introduced into the YOLACT framework: 1) GhostNet for computational efficiency: the ResNet backbone was replaced with GhostNet, which employs Ghost modules (Han <italic>et al.</italic>, 2020). These modules use a combination of base convolutions and cost-effective linear transformations to generate feature maps with reduced redundancy. This architecture decreases the parameter count to 5.2 M (1/5 of ResNet-50) and FLOPs to 0.4 G (1/9 of ResNet-50), while maintaining competitive feature representation. The superior semantic consistency of its multi-scale features facilitates faster inference, making the model suitable for low-power devices. 2) EMAttention for feature enhancement: the dual branch structure of “ECA channel attention and spatial attention” (Ouyang <italic>et al.</italic>, 2023). The ECA (Efficient Channel Attention) channel attention branch employs adaptive 1D convolution to efficiently capture cross-channel interactions with a computational cost only 1/8 of the SE (Squeeze-and-Excitation) module. The spatial attention branch uses depth-wise convolution to generate pixel-wise weight maps, effectively suppressing irrelevant background clutter. This module is designed to improve mask accuracy in challenging scenarios, such as occlusions, with minimal impact on model size and inference speed.</p>
                <sec id="h0-2-1--lightweight-backbone">
                    <title>2.1. Lightweight backbone</title>
                    <p>The original ResNet backbone is replaced with GhostNet, incorporating Ghost modules to produce redundant feature maps while reducing convolutional computation. Specifically, GhostNet employs a two-pronged approach: a 1×1 convolution generates a subset of intrinsic feature maps, which are subsequently expanded via cheap linear operations, such as depth-wise separable convolutions, to increase channel dimensions significantly, and with markedly fewer parameters. The 
                        <italic>Φ
                            <sub>i</sub>
                        </italic> is the identity mapping for preserving the intrinsic feature maps, as shown in Figure 2 (Han <italic>et al.</italic>, 2020).</p>
                    <p>
                        <fig>
                            <caption>
                                <title>Figure 2. Schematic diagram of the GhostNet architecture. (The 
                                    <italic>Φ
                                        <sub>i</sub>
                                    </italic> is the identity mapping for preserving the intrinsic feature maps.)</title>
                            </caption>
                            <graphic xlink:href="media/image2.jpg"/>
                        </fig>
                    </p>
                </sec>
                <sec id="h3-2-2--multi-scale-attention-enhancement">
                    <title>2.2. Multi-scale attention enhancement</title>
                    <p>An EMAttention module is embedded at the skip connection between the Mask Head and the FPN, as depicted in Figure 3 (Ouyang <italic>et al.</italic>, 2023). This module dynamically calibrates channel weights during multi-scale feature fusion through group shuffle and cross-dimensional interactions, thus refining target edges and fine-grained details.</p>
                    <p>
                        <fig>
                            <caption>
                                <title>Figure 3. Schematic diagram of the EMAttention module structure.</title>
                            </caption>
                            <graphic xlink:href="media/image3.png">
                                <alt-text>C:\Users\BLX\xwechat_files\wxid_ltq3oe6i824722_0b18\temp\RWTemp\2026-01\23a6ececc79d3f2ce4f44bd3db06fb41\dc772ad93050184c0f70257801fe441f.png</alt-text>
                            </graphic>
                        </fig>
                    </p>
                    <p>Given an input feature map 
                        <italic>F</italic> with dimensions 
                        <italic>C</italic> × 
                        <italic>H</italic> × 
                        <italic>W</italic>, where 
                        <italic>C</italic> denotes the number of channels, and 
                        <italic>H</italic> and 
                        <italic>W</italic> represent the height and width, respectively, 
                        <italic>G</italic> means the divided groups, the attention mechanism functions as follows:</p>
                    <p>(1) channel attention: channel descriptors are produced using global average pooling and global max pooling. These descriptors are passed through fully connected layers, and channel-wise weights are found using a sigmoid activation function.</p>
                    <p>(2) spatial attention: pooling is conducted across the channel dimension to produce spatial feature maps, which are then processed by convolutional layers to generate spatial weight maps, followed by sigmoid activation. After each training step, the feature map weights are updated using Exponential Moving Average (EMA), computed as shown in Equation 1:</p>
                    <p>
                        <inline-formula>
                            <mml:math>
                                <mml:msub>
                                    <mml:mrow>
                                        <mml:mi>W</mml:mi>
                                    </mml:mrow>
                                    <mml:mrow>
                                        <mml:mi>n</mml:mi>
                                        <mml:mi>e</mml:mi>
                                        <mml:mi>w</mml:mi>
                                    </mml:mrow>
                                </mml:msub>
                                <mml:mi> </mml:mi>
                                <mml:mo>=</mml:mo>
                                <mml:mi> </mml:mi>
                                <mml:mi>α</mml:mi>
                                <mml:msub>
                                    <mml:mrow>
                                        <mml:mi>W</mml:mi>
                                    </mml:mrow>
                                    <mml:mrow>
                                        <mml:mi>o</mml:mi>
                                        <mml:mi>l</mml:mi>
                                        <mml:mi>d</mml:mi>
                                    </mml:mrow>
                                </mml:msub>
                                <mml:mi> </mml:mi>
                                <mml:mo>+</mml:mo>
                                <mml:mi> </mml:mi>
                                <mml:mfenced separators="|">
                                    <mml:mrow>
                                        <mml:mn>1</mml:mn>
                                        <mml:mo>-</mml:mo>
                                        <mml:mi>α</mml:mi>
                                    </mml:mrow>
                                </mml:mfenced>
                                <mml:mi>W</mml:mi>
                                <mml:mi> </mml:mi>
                                <mml:mfenced separators="|">
                                    <mml:mrow>
                                        <mml:mi>E</mml:mi>
                                        <mml:mi>q</mml:mi>
                                        <mml:mo>.</mml:mo>
                                        <mml:mi> </mml:mi>
                                        <mml:mn>1</mml:mn>
                                    </mml:mrow>
                                </mml:mfenced>
                            </mml:math>
                        </inline-formula>
                    </p>
                    <p>where 
                        <italic>W</italic>
                        <sub>new</sub> is the updated attention weight, 
                        <italic>W</italic>
                        <sub>old</sub> signifies the attention weight at the previous moment, 
                        <italic>W</italic> represents the attention weight at the current moment, and 
                        <italic>α</italic> is the decay factor, typically ranging from 0.9 to 0.99. The attention weights are applied to the feature map 
                        <italic>F</italic>
                        <sup> </sup>as shown in Equation 2. The output 
                        <italic>F'</italic> is the weighted feature map 
                        <italic>F</italic>, which is then passed to subsequent network layers for further processing.</p>
                    <p>
                        <inline-formula>
                            <mml:math>
                                <mml:mi>F</mml:mi>
                                <mml:mi mathvariant="normal">'</mml:mi>
                                <mml:mi mathvariant="normal"> </mml:mi>
                                <mml:mo>=</mml:mo>
                                <mml:mi mathvariant="normal"> </mml:mi>
                                <mml:msub>
                                    <mml:mrow>
                                        <mml:mi>F</mml:mi>
                                        <mml:mi>I</mml:mi>
                                    </mml:mrow>
                                    <mml:mrow>
                                        <mml:mi>C</mml:mi>
                                        <mml:mi>W</mml:mi>
                                    </mml:mrow>
                                </mml:msub>
                                <mml:msub>
                                    <mml:mrow>
                                        <mml:mi>I</mml:mi>
                                    </mml:mrow>
                                    <mml:mrow>
                                        <mml:mi>S</mml:mi>
                                        <mml:mi>W</mml:mi>
                                    </mml:mrow>
                                </mml:msub>
                                <mml:mi> </mml:mi>
                                <mml:mfenced separators="|">
                                    <mml:mrow>
                                        <mml:mi>E</mml:mi>
                                        <mml:mi>q</mml:mi>
                                        <mml:mo>.</mml:mo>
                                        <mml:mi> </mml:mi>
                                        <mml:mn>2</mml:mn>
                                    </mml:mrow>
                                </mml:mfenced>
                            </mml:math>
                        </inline-formula>
                    </p>
                    <p>where 
                        <italic>I
                            <sub>CW</sub>
                        </italic> and 
                        <italic>I
                            <sub>SW</sub>
                        </italic> are the channel weight and spatial weight, respectively.</p>
                    <p>The EMA module employs parallel substructures and a shared 1×1 convolutional branch within the CA (Coordinate Attention) module to mitigate extensive sequential processing and excessive network depth. For aggregating multi-scale spatial structural information, EMA strategically places a 3×3 convolutional kernel in parallel with the 1×1 branch, effectively establishing both short-range and long-range dependencies to enhance performance.</p>
                    <p>The improved model is visualised in Figure 4. The model takes an image of size 550 × 550 as input. First, the GhostNet backbone extracts multi-scale base feature maps (denoted as C1–C5) at different down-sampling stages. Subsequently, a feature pyramid network (FPN) performs up-sampling, fusion, and enhancement on C3–C5 to generate integrated feature maps (denoted as P3–P7) that combine rich semantic information with fine spatial details. The model accomplishes the task through two parallel branches: the detection head produces classification scores and bounding box regressions based on P3–P7, while simultaneously predicting a set of 32 mask coefficients for each instance; the prototype mask head (Protonet) generates 32 globally shared prototype masks from P3–P7. The final instance mask is obtained via a linear combination of the prototype masks and the corresponding instance-specific mask coefficients. The results are then post-processed through non-maximum suppression (NMS), mask cropping, and up-sampling to yield pixel-wise instance segmentation masks aligned with the input resolution. The Ghost module generates additional feature maps through inexpensive operations, thereby markedly reducing the model’s computational load. The Ghost module produces rich feature representations while keeping computational costs low, thus improving the performance of small-object detection and segmentation. Moreover, GhostNet’s lightweight nature makes it more suitable for mobile devices and edge computing scenarios, thereby improving the model inference speed and facilitating practical applications.</p>
                    <p>
                        <fig>
                            <caption>
                                <title>Figure 4. Network topology of the improved model.</title>
                            </caption>
                            <graphic xlink:href="media/image4.jpeg">
                                <alt-text>C:\Users\BLX\xwechat_files\wxid_ltq3oe6i824722_0b18\temp\RWTemp\2026-01\124d1856bef625a8ff4d5f6c19fff59d.jpg</alt-text>
                            </graphic>
                        </fig>
                    </p>
                    <p>The EMAttention module is introduced at the skip connection between the Mask Head and the FPN, concatenating multi-scale FPN features with Mask Head features. Spatial convolutions generate spatial attention maps to refine target region features. Global pooling and multilayer perceptrons generate channel attention maps to highlight important channels. Furthermore, adaptive weighted fusion of multiscale features improves segmentation capability for differently sized objects.</p>
                </sec>
            </sec>
            <sec id="h2-3--model-training">
                <title>3. Model training</title>
                <p>The experimental hardware platform was a Dell workstation equipped with an Intel Xeon E5-1620 processor, 32 GB of RAM, and an Nvidia GeForce RTX 2080 Ti graphics card. The operating system was Ubuntu 18.04, and the deep learning framework was PyTorch 1.2 with Python 3.6. The specific model training parameters are detailed in Table 1.</p>
                <table-wrap orientation="portrait" position="float">
                    <caption>
                        <title>Table 1. Model training parameters.</title>
                    </caption>
                    <table>
                        <tbody>
                            <tr>
                                <td valign="middle">
                                    <p>
                                        <bold>Parameter category</bold>
                                    </p>
</td>
                                <td valign="middle">
                                    <p>
                                        <bold>Parameter setting</bold>
                                    </p>
</td>
                            </tr>
                            <tr>
                                <td valign="middle">
                                    <p>Initial learning rate</p>
</td>
                                <td valign="middle">
                                    <p>2e-3</p>
</td>
                            </tr>
                            <tr>
                                <td valign="middle">
                                    <p>Weight decay rate</p>
</td>
                                <td valign="middle">
                                    <p>5e-4</p>
</td>
                            </tr>
                            <tr>
                                <td valign="middle">
                                    <p>Optimizer</p>
</td>
                                <td valign="middle">
                                    <p>SGD</p>
</td>
                            </tr>
                            <tr>
                                <td valign="middle">
                                    <p>Number of epochs</p>
</td>
                                <td valign="middle">
                                    <p>200</p>
</td>
                            </tr>
                        </tbody>
                    </table>
                </table-wrap>
            </sec>
            <sec id="h2-4--evaluation-metrics">
                <title>4. Evaluation metrics</title>
                <p>The evaluation metrics used to assess the model’s performance were precision 
                    <italic>(P)</italic>, recall 
                    <italic>(R)</italic>, average precision 
                    <italic>(AP)</italic>, mean average precision 
                    <italic>(mAP)</italic>, parameter count, and detection frame rate. Precision is the proportion of true-positive detections among all predicted positive instances, serving as an indicator of the model’s accuracy in identifying relevant objects. Recall represents the proportion of true-positive detections among all actual positive instances, reflecting the model’s ability to detect all relevant objects. 
                    <italic>AP</italic> reflects the model’s performance across different recall levels, computed by integrating the precision–recall curve 
                    <italic>(P-R)</italic>. 
                    <italic>mAP</italic> is the mean of 
                    <italic>AP</italic> values across all object classes, offering an overall assessment of the model’s detection capabilities. Additionally, the parameter count and detection frame rate reflect the model’s computational efficiency and real-time processing capability. These metrics collectively provide a holistic view of the model’s effectiveness in object detection tasks. The evaluation metrics were calculated according to Equations 3–6.</p>
                <p>
                    <inline-formula>
                        <mml:math>
                            <mml:mi>P</mml:mi>
                            <mml:mi> </mml:mi>
                            <mml:mo>=</mml:mo>
                            <mml:mi> </mml:mi>
                            <mml:mfrac>
                                <mml:mrow>
                                    <mml:msub>
                                        <mml:mrow>
                                            <mml:mi>T</mml:mi>
                                        </mml:mrow>
                                        <mml:mrow>
                                            <mml:mi>P</mml:mi>
                                        </mml:mrow>
                                    </mml:msub>
                                </mml:mrow>
                                <mml:mrow>
                                    <mml:msub>
                                        <mml:mrow>
                                            <mml:mi>T</mml:mi>
                                        </mml:mrow>
                                        <mml:mrow>
                                            <mml:mi>P</mml:mi>
                                        </mml:mrow>
                                    </mml:msub>
                                    <mml:mi> </mml:mi>
                                    <mml:mo>+</mml:mo>
                                    <mml:mi> </mml:mi>
                                    <mml:msub>
                                        <mml:mrow>
                                            <mml:mi>F</mml:mi>
                                        </mml:mrow>
                                        <mml:mrow>
                                            <mml:mi>P</mml:mi>
                                        </mml:mrow>
                                    </mml:msub>
                                </mml:mrow>
                            </mml:mfrac>
                            <mml:mi> </mml:mi>
                            <mml:mfenced separators="|">
                                <mml:mrow>
                                    <mml:mi>E</mml:mi>
                                    <mml:mi>q</mml:mi>
                                    <mml:mo>.</mml:mo>
                                    <mml:mi> </mml:mi>
                                    <mml:mn>3</mml:mn>
                                </mml:mrow>
                            </mml:mfenced>
                        </mml:math>
                    </inline-formula>
                </p>
                <p>
                    <inline-formula>
                        <mml:math>
                            <mml:mi>R</mml:mi>
                            <mml:mi> </mml:mi>
                            <mml:mo>=</mml:mo>
                            <mml:mi> </mml:mi>
                            <mml:mfrac>
                                <mml:mrow>
                                    <mml:msub>
                                        <mml:mrow>
                                            <mml:mi>T</mml:mi>
                                        </mml:mrow>
                                        <mml:mrow>
                                            <mml:mi>P</mml:mi>
                                        </mml:mrow>
                                    </mml:msub>
                                </mml:mrow>
                                <mml:mrow>
                                    <mml:msub>
                                        <mml:mrow>
                                            <mml:mi>T</mml:mi>
                                        </mml:mrow>
                                        <mml:mrow>
                                            <mml:mi>P</mml:mi>
                                        </mml:mrow>
                                    </mml:msub>
                                    <mml:mi> </mml:mi>
                                    <mml:mo>+</mml:mo>
                                    <mml:mi> </mml:mi>
                                    <mml:msub>
                                        <mml:mrow>
                                            <mml:mi>F</mml:mi>
                                        </mml:mrow>
                                        <mml:mrow>
                                            <mml:mi>N</mml:mi>
                                        </mml:mrow>
                                    </mml:msub>
                                </mml:mrow>
                            </mml:mfrac>
                            <mml:mi> </mml:mi>
                            <mml:mfenced separators="|">
                                <mml:mrow>
                                    <mml:mi>E</mml:mi>
                                    <mml:mi>q</mml:mi>
                                    <mml:mo>.</mml:mo>
                                    <mml:mi> </mml:mi>
                                    <mml:mn>4</mml:mn>
                                </mml:mrow>
                            </mml:mfenced>
                        </mml:math>
                    </inline-formula>
                </p>
                <p>
                    <inline-formula>
                        <mml:math>
                            <mml:mi>A</mml:mi>
                            <mml:mi>P</mml:mi>
                            <mml:mi> </mml:mi>
                            <mml:mo>=</mml:mo>
                            <mml:mi> </mml:mi>
                            <mml:mrow>
                                <mml:msubsup>
                                    <mml:mo stretchy="false">∫</mml:mo>
                                    <mml:mrow>
                                        <mml:mn>0</mml:mn>
                                    </mml:mrow>
                                    <mml:mrow>
                                        <mml:mn>1</mml:mn>
                                    </mml:mrow>
                                </mml:msubsup>
                                <mml:mrow>
                                    <mml:mi>P</mml:mi>
                                    <mml:mfenced separators="|">
                                        <mml:mrow>
                                            <mml:mi>R</mml:mi>
                                        </mml:mrow>
                                    </mml:mfenced>
                                    <mml:mi> </mml:mi>
                                    <mml:mi>d</mml:mi>
                                    <mml:mi>R</mml:mi>
                                </mml:mrow>
                            </mml:mrow>
                            <mml:mi> </mml:mi>
                            <mml:mfenced separators="|">
                                <mml:mrow>
                                    <mml:mi>E</mml:mi>
                                    <mml:mi>q</mml:mi>
                                    <mml:mo>.</mml:mo>
                                    <mml:mi> </mml:mi>
                                    <mml:mn>5</mml:mn>
                                </mml:mrow>
                            </mml:mfenced>
                        </mml:math>
                    </inline-formula>
                </p>
                <p>
                    <inline-formula>
                        <mml:math>
                            <mml:mi>m</mml:mi>
                            <mml:mi>A</mml:mi>
                            <mml:mi>P</mml:mi>
                            <mml:mi> </mml:mi>
                            <mml:mo>=</mml:mo>
                            <mml:mi> </mml:mi>
                            <mml:mfrac>
                                <mml:mrow>
                                    <mml:mrow>
                                        <mml:msubsup>
                                            <mml:mo stretchy="false">∑</mml:mo>
                                            <mml:mrow>
                                                <mml:mi>i</mml:mi>
                                                <mml:mi> </mml:mi>
                                                <mml:mo>=</mml:mo>
                                                <mml:mi> </mml:mi>
                                                <mml:mn>1</mml:mn>
                                            </mml:mrow>
                                            <mml:mrow>
                                                <mml:mi>l</mml:mi>
                                            </mml:mrow>
                                        </mml:msubsup>
                                        <mml:mrow>
                                            <mml:msub>
                                                <mml:mrow>
                                                    <mml:mi>A</mml:mi>
                                                    <mml:mi>P</mml:mi>
                                                </mml:mrow>
                                                <mml:mrow>
                                                    <mml:mi>i</mml:mi>
                                                </mml:mrow>
                                            </mml:msub>
                                        </mml:mrow>
                                    </mml:mrow>
                                </mml:mrow>
                                <mml:mrow>
                                    <mml:mi>l</mml:mi>
                                </mml:mrow>
                            </mml:mfrac>
                            <mml:mi> </mml:mi>
                            <mml:mfenced separators="|">
                                <mml:mrow>
                                    <mml:mi>E</mml:mi>
                                    <mml:mi>q</mml:mi>
                                    <mml:mo>.</mml:mo>
                                    <mml:mi> </mml:mi>
                                    <mml:mn>6</mml:mn>
                                </mml:mrow>
                            </mml:mfenced>
                        </mml:math>
                    </inline-formula>
                </p>
                <p>In the formula, 
                    <italic>T
                        <sub>P</sub>
                    </italic> denotes the number of true-positive detections; 
                    <italic>F
                        <sub>P</sub>
                    </italic> represents the number of false-positive detections; 
                    <italic>F
                        <sub>N</sub>
                    </italic> signifies the number of false-negative detections; 
                    <italic>l</italic> refers to the number of detection categories; and 
                    <italic>AP</italic> is the area under the 
                    <italic>P-R</italic> curve, where 
                    <italic>R</italic> is plotted on the 
                    <italic>x</italic>-axis, and 
                    <italic>P</italic> is shown on the 
                    <italic>y</italic>-axis. 
                    <italic>AP</italic> reflects a comprehensive measure of model performance.</p>
                <p>For automated pruning and yield estimation, minimising false negatives (
                    <italic>i.e.</italic>, maximising recall) is often more critical. In dormant pruning, missing a shoot that requires removal 
                    <italic>(F
                        <sub>N</sub>)</italic> means it will be left on the vine, potentially leading to incorrect canopy structure, wasted nutrients, and compromised yield/quality in the following season. Similarly, for yield estimation, failing to count a shoot 
                    <italic>(F
                        <sub>N</sub>)</italic> directly leads to an underestimation of the potential yield, preventing accurate management decisions. Therefore, a high recall ensures that most management targets are captured. For generating precise maps or for robot navigation, minimising false positives (
                    <italic>i.e.</italic>, maximising precision) can be more important. If the model mistakes shadows or weeds for shoots 
                    <italic>(F
                        <sub>P</sub>)</italic>, a canopy map generated from this information would be noisy and misleading for management. For a physical robot, attempting to cut a non-existent “shoot” is not only an inefficient action but also poses a risk of mechanical failure and energy waste. In this research, the average precision 
                    <italic>(AP)</italic> was chosen as a core metric precisely because it integrates both precision and recall across multiple confidence thresholds (
                    <italic>i.e.</italic>, the area under the precision-recall curve). A high 
                    <italic>mAP</italic> score indicates that the model achieves a good overall balance between “finding all relevant objects (high recall)” and “ensuring the detected objects are correct (high precision)”, which is most desirable for a versatile automation system expected to perform multiple tasks.</p>
            </sec>
        </sec>
        <sec id="h1-results-and-discussion">
            <title>Results and discussion</title>
            <p>To evaluate the impact of different attention mechanisms on the YOLACT model, the study incorporated Squeeze-and-Excitation (SE) (Hu <italic>et al.</italic>, 2018), Coordinate Attention (CA) (Hou <italic>et al.</italic>, 2021), Efficient Channel Attention (ECA) (Wang <italic>et al.</italic>, 2020), and EMAttention modules into the YOLACT architecture. The comparative experimental results are given in Table 2. As shown, the introduction of different attention modules led to a certain degree of improvement in the segmentation performance of the original model. Of these models, the EMAttention module exhibited the greatest improvement, with 
                <italic>P</italic>, 
                <italic>R</italic>, and 
                <italic>mAP</italic> increasing to 74.5 %, 64.2 %, and 67.98 %, respectively.</p>
            <table-wrap orientation="portrait" position="float">
                <caption>
                    <title>Table 2. Comparative experiment of integrating different attention modules into the YOLACT network.</title>
                </caption>
                <table>
                    <tbody>
                        <tr>
                            <td valign="middle">
                                <p>
                                    <bold>Attention modules</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>
                                        <italic>P</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>
                                        <italic>R</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>
                                        <italic>mAP</italic>/%</bold>
                                </p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>YOLACT</p>
</td>
                            <td valign="middle">
                                <p>72.1</p>
</td>
                            <td valign="middle">
                                <p>63.3</p>
</td>
                            <td valign="middle">
                                <p>64.83</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>SE</p>
</td>
                            <td valign="middle">
                                <p>72.9</p>
</td>
                            <td valign="middle">
                                <p>64.8</p>
</td>
                            <td valign="middle">
                                <p>65.32</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>CA</p>
</td>
                            <td valign="middle">
                                <p>73.3</p>
</td>
                            <td valign="middle">
                                <p>63.9</p>
</td>
                            <td valign="middle">
                                <p>65.67</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>ECA</p>
</td>
                            <td valign="middle">
                                <p>73.2</p>
</td>
                            <td valign="middle">
                                <p>64.0</p>
</td>
                            <td valign="middle">
                                <p>66.03</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>EMAttention</p>
</td>
                            <td valign="middle">
                                <p>74.5</p>
</td>
                            <td valign="middle">
                                <p>64.2</p>
</td>
                            <td valign="middle">
                                <p>67.98</p>
</td>
                        </tr>
                    </tbody>
                </table>
            </table-wrap>
            <p>To evaluate the impact of different lightweight backbone networks on the segmentation and detection performance of the model, this study replaced the original backbone with MobileNetV2 (Sandler <italic>et al.</italic>, 2018), MobileNetV3 (Howard <italic>et al.</italic>, 2019), ShuffleNetV2 (Ma <italic>et al.</italic>, 2018), and GhostNet. The comparative experimental results are presented in Table 3. As shown, the GhostNet backbone achieves the best performance across all metrics, with 
                <italic>P</italic>, 
                <italic>R</italic>, 
                <italic>mAP</italic>, and parameter count improving by 1.7, 1.9, and 1.04 percentage points, respectively, compared with the original model. Additionally, the model’s parameter count was reduced to 35.64 MB, or a decrease of 15.78 %.</p>
            <table-wrap orientation="portrait" position="float">
                <caption>
                    <title>Table 3. Comparative experiment of the YOLACT network with different backbone networks.</title>
                </caption>
                <table>
                    <tbody>
                        <tr>
                            <td valign="middle">
                                <p>
                                    <bold>Backbone networks</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>
                                        <italic>P</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>
                                        <italic>R</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>
                                        <italic>mAP</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Model parameter/MB</bold>
                                </p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>MobileNetV2</p>
</td>
                            <td valign="middle">
                                <p>72.5</p>
</td>
                            <td valign="middle">
                                <p>63.9</p>
</td>
                            <td valign="middle">
                                <p>64.88</p>
</td>
                            <td valign="middle">
                                <p>36.75</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>MobileNetV3</p>
</td>
                            <td valign="middle">
                                <p>73.0</p>
</td>
                            <td valign="middle">
                                <p>64.8</p>
</td>
                            <td valign="middle">
                                <p>65.27</p>
</td>
                            <td valign="middle">
                                <p>38.66</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>ShuffleNetV2</p>
</td>
                            <td valign="middle">
                                <p>73.3</p>
</td>
                            <td valign="middle">
                                <p>64.6</p>
</td>
                            <td valign="middle">
                                <p>65.31</p>
</td>
                            <td valign="middle">
                                <p>39.06</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>GhostNet</p>
</td>
                            <td valign="middle">
                                <p>73.8</p>
</td>
                            <td valign="middle">
                                <p>65.2</p>
</td>
                            <td valign="middle">
                                <p>65.87</p>
</td>
                            <td valign="middle">
                                <p>35.64</p>
</td>
                        </tr>
                    </tbody>
                </table>
            </table-wrap>
            <p>Figure 5 illustrates the loss curves during training for four models: YOLACT, YOLACT + GhostNet, YOLACT + EMAttention, and YOLACT + GhostNet + EMAttention. The curves reveal that the YOLACT + GhostNet + EMAttention model converged more rapidly and achieved the lowest loss on the test set compared with the other models. Table 4 lists the performance metrics of these models, demonstrating that the proposed modifications improved the segmentation performance. In order to avoid the randomness of the results caused by a single training, each training is repeated three times, and the average and standard deviation of the three training results are calculated. Specifically, the improved model achieved a mask 
                <italic>mAP</italic> of 67.83 ± 0.61 %, signifying a 4.00 percentage point increase over the original model. The detection 
                <italic>mAP</italic> was 68.57 ± 0.74 %, or a 3.75 percentage point improvement. These results underscore the effectiveness of integrating GhostNet and EMAttention modules in augmenting the segmentation capabilities for grapevine and shoot instances. Figure 6 provides a visual comparison of segmentation results for grapevine and shoots across the four models. The enhanced model demonstrates superior segmentation accuracy, particularly in delineating fine structures and small targets. Furthermore, the optimised model operated at a detection speed of 46.72 frames per second (FPS), representing a 17.59 % increase over the original model. This improvement aligns with the goal of achieving both higher segmentation quality and a faster detection speed.</p>
            <p>
                <fig>
                    <caption>
                        <title>Figure 5. Loss curves during the training process of four models.</title>
                    </caption>
                    <graphic xlink:href="media/image5.jpg"/>
                </fig>
            </p>
            <p>
                <fig>
                    <caption>
                        <title>Figure 6. Comparison of grapevine segmentation results across four models.</title>
                        <p>Note: the yellow boxes indicate areas where the original model failed to accurately segment the grapevine and side branches, whereas the improved model achieved precise segmentation.</p>
                    </caption>
                    <graphic xlink:href="media/image6.jpg"/>
                </fig>
            </p>
            <p/>
            <table-wrap orientation="portrait" position="float">
                <caption>
                    <title>Table 4. Ablation test results.</title>
                </caption>
                <table>
                    <tbody>
                        <tr>
                            <td valign="middle">
                                <p>
                                    <bold>Model</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>
                                        <italic>P</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>
                                        <italic>R</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Bounding box 
                                        <italic>mAP</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Bounding box 
                                        <italic>AP</italic>75/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Mask 
                                        <italic>mAP</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Mask 
                                        <italic>AP</italic>75/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Detection speed/FPS</bold>
                                </p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>YOLACT</p>
</td>
                            <td valign="middle">
                                <p>72.13 ± 1.35</p>
</td>
                            <td valign="middle">
                                <p>63.20 ± 0.36</p>
</td>
                            <td valign="middle">
                                <p>64.82 ± 0.06</p>
</td>
                            <td valign="middle">
                                <p>75.51 ± 0.03</p>
</td>
                            <td valign="middle">
                                <p>63.83 ± 0.25</p>
</td>
                            <td valign="middle">
                                <p>71.04 ± 0.20</p>
</td>
                            <td valign="middle">
                                <p>39.73</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>YOLACT + GhostNet</p>
</td>
                            <td valign="middle">
                                <p>73.43 ± 0.72</p>
</td>
                            <td valign="middle">
                                <p>64.17 ± 1.17</p>
</td>
                            <td valign="middle">
                                <p>65.88 ± 0.62</p>
</td>
                            <td valign="middle">
                                <p>77.73 ± 0.11</p>
</td>
                            <td valign="middle">
                                <p>65.25 ± 0.30</p>
</td>
                            <td valign="middle">
                                <p>72.95 ± 1.69</p>
</td>
                            <td valign="middle">
                                <p>47.85</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>YOLACT + EMAttention</p>
</td>
                            <td valign="middle">
                                <p>73.30 ± 1.82</p>
</td>
                            <td valign="middle">
                                <p>64.73 ± 0.68</p>
</td>
                            <td valign="middle">
                                <p>67.31 ± 0.94</p>
</td>
                            <td valign="middle">
                                <p>78.27 ± 0.09</p>
</td>
                            <td valign="middle">
                                <p>63.78 ± 0.35</p>
</td>
                            <td valign="middle">
                                <p>70.82 ± 0.69</p>
</td>
                            <td valign="middle">
                                <p>38.77</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>YOLACT + GhostNet + EMAttention</p>
</td>
                            <td valign="middle">
                                <p>75.70 ± 0.75</p>
</td>
                            <td valign="middle">
                                <p>66.5 ± 0.75</p>
</td>
                            <td valign="middle">
                                <p>68.57 ± 0.74</p>
</td>
                            <td valign="middle">
                                <p>79.16 ± 0.88</p>
</td>
                            <td valign="middle">
                                <p>67.83 ± 0.61</p>
</td>
                            <td valign="middle">
                                <p>73.60 ± 0.83</p>
</td>
                            <td valign="middle">
                                <p>46.72</p>
</td>
                        </tr>
                    </tbody>
                </table>
            </table-wrap>
            <p>To evaluate the performance of the proposed model on grape images, we conducted comparative experiments against several mainstream segmentation models to assess its effectiveness and generalisation capability. Specifically, the study selected Mask R-CNN, YOLACT++, BlendMask, and SOLOv2 for comparison. As shown in Table 5, the proposed model demonstrated significant advantages in both mask 
                <italic>mAP</italic> and bounding box, achieving 67.83 ± 0.61 % and 68.57 ± 0.74 %, respectively. The detection accuracy was 75.7 ± 0.75 %, surpassing that of the other models. Although the 
                <italic>R</italic> of the improved model was 2.15 percentage points lower than that of BlendMask, it outperformed Mask R-CNN, YOLACT++, and SOLOv2 by 3.45, 2.42, and 1.37 percentage points, respectively. Furthermore, the bounding box 
                <italic>mAP</italic> of the improved model exceeded that of Mask R-CNN, YOLACT++, BlendMask, and SOLOv2 by 4.88, 3.89, 3.96, and 3.29 percentage points, respectively. In terms of detection speed, the model operated at 46.72 FPS, representing improvements of 96.88 %, 17.59 %, 22.66 %, and 8.52 % over Mask R-CNN, YOLACT++, BlendMask, and SOLOv2, respectively. Thus, the proposed model not only improved segmentation performance but also achieved higher detection speed, indicating its suitability for real-time applications in grapevine image analysis. SOLOv2 model is superior to other models in terms of 
                <italic>P</italic>, 
                <italic>mAP</italic> and detection speed. The YOLACT++ model is improved by introducing the GhostNet and EMAttention module. Therefore, the improved model has a better detection effect.</p>
            <table-wrap orientation="portrait" position="float">
                <caption>
                    <title>Table 5. Comparative performance of five models in the grapevine image.</title>
                </caption>
                <table>
                    <tbody>
                        <tr>
                            <td valign="middle">
                                <p>
                                    <bold>Model</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>
                                        <italic>P</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>
                                        <italic>R</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Bounding box 
                                        <italic>mAP</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Bounding box 
                                        <italic>AP</italic>75/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Mask 
                                        <italic>mAP</italic>/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Mask 
                                        <italic>AP</italic>75/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Detection speed/FPS</bold>
                                </p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>Mask R-CNN</p>
</td>
                            <td valign="middle">
                                <p>68.45 ± 0.85</p>
</td>
                            <td valign="middle">
                                <p>63.05 ± 0.13</p>
</td>
                            <td valign="middle">
                                <p>63.69 ± 0.19</p>
</td>
                            <td valign="middle">
                                <p>74.64 ± 0.98</p>
</td>
                            <td valign="middle">
                                <p>62.94 ± 1.17</p>
</td>
                            <td valign="middle">
                                <p>69.47 ± 1.40</p>
</td>
                            <td valign="middle">
                                <p>23.73</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>YOLACT++</p>
</td>
                            <td valign="middle">
                                <p>73.52 ± 0.38</p>
</td>
                            <td valign="middle">
                                <p>64.08 ± 0.25</p>
</td>
                            <td valign="middle">
                                <p>64.68 ± 0.64</p>
</td>
                            <td valign="middle">
                                <p>75.62 ± 0.45</p>
</td>
                            <td valign="middle">
                                <p>63.62 ± 1.00</p>
</td>
                            <td valign="middle">
                                <p>70.90 ± 0.59</p>
</td>
                            <td valign="middle">
                                <p>39.73</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>BlendMask</p>
</td>
                            <td valign="middle">
                                <p>69.35 ± 0.38</p>
</td>
                            <td valign="middle">
                                <p>68.65 ± 0.49</p>
</td>
                            <td valign="middle">
                                <p>64.61 ± 0.29</p>
</td>
                            <td valign="middle">
                                <p>76.85 ± 0.55</p>
</td>
                            <td valign="middle">
                                <p>60.95 ± 0.35</p>
</td>
                            <td valign="middle">
                                <p>65.99 ± 2.51</p>
</td>
                            <td valign="middle">
                                <p>38.09</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>SOLOv2</p>
</td>
                            <td valign="middle">
                                <p>74.77 ± 0.40</p>
</td>
                            <td valign="middle">
                                <p>65.13 ± 0.25</p>
</td>
                            <td valign="middle">
                                <p>65.28 ± 1.65</p>
</td>
                            <td valign="middle">
                                <p>77.53 ± 0.19</p>
</td>
                            <td valign="middle">
                                <p>65.51 ± 0.85</p>
</td>
                            <td valign="middle">
                                <p>70.47 ± 0.27</p>
</td>
                            <td valign="middle">
                                <p>43.05</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>Ours</p>
</td>
                            <td valign="middle">
                                <p>75.70 ± 0.75</p>
</td>
                            <td valign="middle">
                                <p>66.50 ± 0.75</p>
</td>
                            <td valign="middle">
                                <p>68.57 ± 0.74</p>
</td>
                            <td valign="middle">
                                <p>79.16 ± 0.88</p>
</td>
                            <td valign="middle">
                                <p>67.83 ± 0.61</p>
</td>
                            <td valign="middle">
                                <p>73.60 ± 0.83</p>
</td>
                            <td valign="middle">
                                <p>46.72</p>
</td>
                        </tr>
                    </tbody>
                </table>
            </table-wrap>
            <p>To validate the performance of the improved lightweight model proposed in this study on embedded devices, a Jetson Xavier NX developer kit (memory: 8 GB; GPU: 384-core NVIDIA Volta GPU with 48 Tensor Cores; CPU: 6-core NVIDIA Carmel ARM v8.2) running Ubuntu 22.04 was employed. The segmentation and detection speed of the improved model reached 2.34 FPS.</p>
            <p>As shown in Table 6, the experimental results show that the proposed model achieves values of 69.09 ± 0.89 % for grapevine detection and 68.05 ± 0.72 % for shoot detection. These results represent improvements of 6.03, 4.62, 4.56, and 4.33 percentage points over Mask R-CNN, YOLACT++, BlendMask, and SOLOv2, respectively, in grapevine detection. For side branch detection, the improvements were 3.72, 3.17, 3.36, and 2.25 percentage points over the same models.</p>
            <table-wrap orientation="portrait" position="float">
                <caption>
                    <title>Table 6. Comparative experiment on the average precision of grapevines and shoots.</title>
                </caption>
                <table>
                    <tbody>
                        <tr>
                            <td valign="middle">
                                <p>
                                    <bold>Model</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Grapevines/%</bold>
                                </p>
</td>
                            <td valign="middle">
                                <p>
                                    <bold>Shoots/%</bold>
                                </p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>Mask R-CNN</p>
</td>
                            <td valign="middle">
                                <p>63.06 ± 1.02</p>
</td>
                            <td valign="middle">
                                <p>64.33 ± 0.78</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>YOLACT++</p>
</td>
                            <td valign="middle">
                                <p>64.47 ± 1.93</p>
</td>
                            <td valign="middle">
                                <p>64.88 ± 0.83</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>BlendMask</p>
</td>
                            <td valign="middle">
                                <p>64.53 ± 1.72</p>
</td>
                            <td valign="middle">
                                <p>64.69 ± 1.64</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>SOLOv2</p>
</td>
                            <td valign="middle">
                                <p>64.76 ± 1.44</p>
</td>
                            <td valign="middle">
                                <p>65.80 ± 2.08</p>
</td>
                        </tr>
                        <tr>
                            <td valign="middle">
                                <p>Ours</p>
</td>
                            <td valign="middle">
                                <p>69.09 ± 0.89</p>
</td>
                            <td valign="middle">
                                <p>68.05 ± 0.72</p>
</td>
                        </tr>
                    </tbody>
                </table>
            </table-wrap>
            <p>While the performance gains in detection accuracy and inference speed presented in this study may appear modest in absolute terms, their practical implications for vineyard operations are significant. SOLOv2, while effective for instance segmentation, demands considerable computational resources that can hinder deployment on embedded systems commonly used in agricultural robotics. Our improved YOLACT model achieves a better balance between accuracy and computational efficiency, making it more amenable for real-time applications under typical field conditions. In practice, real-time segmentation in vineyards requires not only high 
                <italic>mAP</italic> but also sustained high inference speed on mid-range hardware to achieve responsive robotic control. Although SOLOv2 can achieve real-time performance on high-end GPUs, its deployment on cost-effective, power-efficient platforms—often necessary for field robots—remains challenging. The proposed model strikes a balance between competitive segmentation accuracy (~69 % 
                <italic>mAP</italic>) and a high inference speed of 46.72 FPS, thereby facilitating its integration into automated vineyard systems. But the practical applicability of the model is currently constrained by the characteristics of the Grapevine-Seg dataset. The annotations were manually completed and, consequently, reflect specific grapevine varieties, trellising systems, and environmental conditions. An inherent class imbalance exists within the dataset, where shoot instances significantly outnumber cordon instances. Furthermore, potential geographic and seasonal biases persist. While data augmentation techniques were employed to mitigate these issues, they may still impact the model’s generalisation capability. The public release of this dataset represents an initial step to address this limitation by encouraging the research community to incorporate more diverse data.</p>
            <p>Furthermore, the ultimate objective of real-time grapevine segmentation extends beyond mere detection; it serves as a critical enabler for precision agriculture tasks such as automated pruning, yield estimation, and canopy management. 1) Automated pruning: the instance masks of shoots and the cordon generated by the model enable precise localisation of their junction points. This provides crucial spatial coordinates for a robotic cutting tool, allowing it to plan optimal cutting paths to selectively remove unwanted shoots while preserving healthy fruiting wood. 2) Shoot counting: the automatic count of segmented shoot instances facilitates the estimation of shoot density per unit length of cordon. This metric serves as a fundamental input for constructing accurate yield prediction models and for informing crop-load management decisions, such as cluster thinning. 3) Canopy management: the segmentation masks allow for the assessment of shoot distribution uniformity. By identifying overcrowded zones, the system can guide targeted shoot thinning operations. This optimises canopy light exposure and air circulation, thereby enhancing final fruit quality. Segmentation is, therefore, a prerequisite for the decision-making process of determining “which cane to cut” and “where to make the cut”. The proposed method’s improved speed-accuracy trade-off ensures that segmentation outputs can be processed within the tight latency constraints of closed-loop robotic control, thereby supporting continuous and adaptive operation in dynamic vineyard environments.</p>
        </sec>
        <sec id="h1-conclusion">
            <title>Conclusion</title>
            <p>In automated viticultural management, precise segmentation of vines and side branches is crucial for tasks such as pruning. To achieve rapid and accurate component segmentation, this study introduced the EMAttention and GhostNet modules into the YOLACT framework, resulting in a lightweight model that increases the detection speed and segmentation accuracy. Comparative experimental analyses yield the following conclusions:</p>
            <p>1. The improved model achieved a bounding box 
                <italic>mAP</italic> of 68.57 ± 0.74 %, mask 
                <italic>mAP</italic> of 67.83 ± 0.61 %, and a detection speed of 46.72 FPS, representing increases of 3.75, 4.00 percentage points, and 17.59 %, respectively, over the original model. The average precision
                <bold> </bold>for grapevines and shoots was 69.09 ± 0.89 % and 68.05 ± 0.72 %, respectively.</p>
            <p>2. The superior performance-efficiency balance achieved by our improved YOLACT-based model. When evaluated under consistent conditions, the proposed method exhibits a compelling advantage over leading segmentation approaches—including Mask R-CNN, YOLACT++, BlendMask, and SOLOv2—by simultaneously elevating detection accuracy (as reflected in bounding box 
                <italic>mAP</italic>), enhancing mask quality, and reducing computational complexity. Notably, the model also delivers the fastest inference speed among all compared frameworks, thereby reinforcing its practical viability for real-time agricultural applications. Although BlendMask attained a marginally higher recall, the comprehensive gains across multiple metrics affirm that our approach offers a more effective and deployable solution for grapevine instance segmentation in resource-conscious field environments.</p>
            <p>It should be noted that this study has certain limitations regarding the training and testing datasets. All images were collected from a single grape variety (Cabernet-Sauvignon), relatively young vines (7–8 years old), and a uniform training system (trellised Cordon de Royat). These constraints may affect the generalisability of the model to other viticultural contexts, as older vines typically exhibit greater structural complexity and different varieties may present distinct morphological characteristics.</p>
            <p>Regarding practical applicability, the segmentation outputs generated by our model can be directly integrated into robotic systems for tasks such as automated pruning or thinning. For example, the instance masks corresponding to individual cane structures can be used by a path planning module to guide a robotic manipulator in making precise cuts during dormant pruning. Similarly, during canopy management, the segmented foliage regions can inform selective thinning operations to optimise sunlight exposure and air circulation. While this study focused on algorithm development, future work will include field validation via integration with a robotic platform equipped with a real-time perception-control loop. Prior to embedded deployment, it would be scientifically prudent to validate the proposed approach across diverse wine-growing regions worldwide, encompassing a wider range of grape varieties, vine ages, and training systems. Specifically, we plan to implement the proposed model on a pruning robot to evaluate its performance in operational scenarios, measuring task completion rates and robustness under varying field conditions.</p>
        </sec>
        <sec id="h1-acknowledgements">
            <title>Acknowledgements</title>
            <p>The work in this paper was supported by the Natural Science Foundation of Ningxia (2025AAC030073, 2023AAC03302), Key Research and Development Project of Ningxia Hui Autonomous Region (2024BEH04137), and North Minzu University (2021KYQD31).</p>
        </sec>
    </body>
    <back>
        <fn-group/>
        <ref-list>
            <ref id="ref1">
                <label>1</label>
                <mixed-citation>
                    <name>
                        <surname>Bochtis</surname>
                        <given-names>D.</given-names>
                    </name>
                    <name>
                        <surname>Sørensen</surname>
                        <given-names>C.</given-names>
                    </name>
                    <name>
                        <surname>Busato</surname>
                        <given-names>P.</given-names>
                    </name>
                    <year>2014</year>
                    <article-title>Advances in agricultural machinery management: A review</article-title>
                    <source>Biosystems Engineering</source>
                    <volume>126</volume>
                    <page-range>69-81</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1016/j.biosystemseng.2014.07.012">https://doi.org/10.1016/j.biosystemseng.2014.07.012</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref2">
                <label>2</label>
                <mixed-citation>
                    <name>
                        <surname>Bolya</surname>
                        <given-names>D.</given-names>
                    </name>
                    <name>
                        <surname>Zhou</surname>
                        <given-names>C.</given-names>
                    </name>
                    <name>
                        <surname>Xiao</surname>
                        <given-names>F.</given-names>
                    </name>
                    <name>
                        <surname>Lee</surname>
                        <given-names>Y.</given-names>
                    </name>
                    <year>2019</year>
                    <article-title>YOLACT: Real-time instance segmentation</article-title>
                    <publisher-name>Paper presented at the 2019 IEEE/CVF International Conference on Computer Vision (ICCV)</publisher-name>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/ICCV.2019.00925">https://doi.org/10.1109/ICCV.2019.00925</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref3">
                <label>3</label>
                <mixed-citation>
                    <name>
                        <surname>Botterill</surname>
                        <given-names>T.</given-names>
                    </name>
                    <name>
                        <surname>Paulin</surname>
                        <given-names>S.</given-names>
                    </name>
                    <name>
                        <surname>Green</surname>
                        <given-names>R.</given-names>
                    </name>
                    <name>
                        <surname>Williams</surname>
                        <given-names>S.</given-names>
                    </name>
                    <name>
                        <surname>Lin</surname>
                        <given-names>J.</given-names>
                    </name>
                    <name>
                        <surname>Saxton</surname>
                        <given-names>V.</given-names>
                    </name>
                    <name>
                        <surname>Mills</surname>
                        <given-names>S.</given-names>
                    </name>
                    <name>
                        <surname>Chen</surname>
                        <given-names>X.</given-names>
                    </name>
                    <name>
                        <surname>Corbett-Davies</surname>
                        <given-names>S.</given-names>
                    </name>
                    <year>2017</year>
                    <article-title>A robot system for pruning grape vines</article-title>
                    <source>Journal of Field Robotics</source>
                    <volume>34</volume>
                    <issue>6</issue>
                    <page-range>1100-1122</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1002/rob.21680">https://doi.org/10.1002/rob.21680</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref4">
                <label>4</label>
                <mixed-citation>
                    <name>
                        <surname>Bu</surname>
                        <given-names>L.</given-names>
                    </name>
                    <name>
                        <surname>Zhang</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <name>
                        <surname>Kou</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <name>
                        <surname>Chen</surname>
                        <given-names>Y.</given-names>
                    </name>
                    <name>
                        <surname>Li</surname>
                        <given-names>X.</given-names>
                    </name>
                    <name>
                        <surname>Wu</surname>
                        <given-names>X.</given-names>
                    </name>
                    <name>
                        <surname>Zhang</surname>
                        <given-names>T.</given-names>
                    </name>
                    <name>
                        <surname>Tang</surname>
                        <given-names>X.</given-names>
                    </name>
                    <name>
                        <surname>Wang</surname>
                        <given-names>J.</given-names>
                    </name>
                    <name>
                        <surname>Zhao</surname>
                        <given-names>L.</given-names>
                    </name>
                    <year>2025</year>
                    <article-title>Investigating shear force and torque of grapevine shoots based on experimental and simulation analysis</article-title>
                    <source>BioResources</source>
                    <volume>20</volume>
                    <issue>3</issue>
                    <page-range>6662-6679</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.15376/biores.20.3.6662-6679">https://doi.org/10.15376/biores.20.3.6662-6679</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref5">
                <label>5</label>
                <mixed-citation>
                    <name>
                        <surname>Casado-García</surname>
                        <given-names>A.</given-names>
                    </name>
                    <name>
                        <surname>Heras</surname>
                        <given-names>J.</given-names>
                    </name>
                    <name>
                        <surname>Milella</surname>
                        <given-names>A.</given-names>
                    </name>
                    <name>
                        <surname>Marani</surname>
                        <given-names>R.</given-names>
                    </name>
                    <year>2022</year>
                    <article-title>Semi-supervised deep learning and low-cost cameras for the semantic segmentation of natural images in viticulture</article-title>
                    <source>Precision Agriculture</source>
                    <volume>23</volume>
                    <issue>6</issue>
                    <page-range>2001-2026</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1007/s11119-022-09929-9">https://doi.org/10.1007/s11119-022-09929-9</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref6">
                <label>6</label>
                <mixed-citation>
                    <name>
                        <surname>Dong</surname>
                        <given-names>Y.</given-names>
                    </name>
                    <name>
                        <surname>Hu</surname>
                        <given-names>G.</given-names>
                    </name>
                    <name>
                        <surname>Liu</surname>
                        <given-names>G.</given-names>
                    </name>
                    <name>
                        <surname>Tohti</surname>
                        <given-names>G.</given-names>
                    </name>
                    <year>2016</year>
                    <article-title>Segmentation method for grapevine critical structure based on Mask R-CNN model</article-title>
                    <source>Journal of Chinese Agricultural Mechanization</source>
                    <volume>45</volume>
                    <issue>2</issue>
                    <page-range>207-214</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.13733/j.jcam.issn.2095-5553.2024.02.030">https://doi.org/10.13733/j.jcam.issn.2095-5553.2024.02.030</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref7">
                <label>7</label>
                <mixed-citation>
                    <name>
                        <surname>Fernandes</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Gamba</surname>
                        <given-names>J. D.</given-names>
                    </name>
                    <name>
                        <surname>Pelusi</surname>
                        <given-names>F.</given-names>
                    </name>
                    <name>
                        <surname>Bratta</surname>
                        <given-names>A.</given-names>
                    </name>
                    <name>
                        <surname>Caldwell</surname>
                        <given-names>D.</given-names>
                    </name>
                    <name>
                        <surname>Poni</surname>
                        <given-names>S.</given-names>
                    </name>
                    <name>
                        <surname>Semini</surname>
                        <given-names>C.</given-names>
                    </name>
                    <year>2025</year>
                    <article-title>Grapevine winter pruning: Merging 2D segmentation and 3D point clouds for pruning point generation</article-title>
                    <source>Computers and Electronics in Agriculture</source>
                    <volume>237</volume>
                    <page-range>110589</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1016/j.compag.2025.110589">https://doi.org/10.1016/j.compag.2025.110589</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref8">
                <label>8</label>
                <mixed-citation>
                    <name>
                        <surname>Gao</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Lu</surname>
                        <given-names>T. F.</given-names>
                    </name>
                    <year>2006</year>
                    <article-title>Image processing and analysis for autonomous grapevine pruning</article-title>
                    <source>International Conference on Mechatronics and Automation</source>
                    <volume>2006</volume>
                    <page-range>922–927</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/ICMA.2006.257748">https://doi.org/10.1109/ICMA.2006.257748</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref9">
                <label>9</label>
                <mixed-citation>
                    <name>
                        <surname>Gentilhomme</surname>
                        <given-names>T.</given-names>
                    </name>
                    <name>
                        <surname>Villamizar</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Corre</surname>
                        <given-names>J.</given-names>
                    </name>
                    <name>
                        <surname>Odobez</surname>
                        <given-names>J.</given-names>
                    </name>
                    <year>2023</year>
                    <article-title>Towards smart pruning: ViNet, a deep-learning approach for grapevine structure estimation</article-title>
                    <source>Computers and Electronics in Agriculture</source>
                    <volume>207</volume>
                    <page-range>107736</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1016/j.compag.2023.107736">https://doi.org/10.1016/j.compag.2023.107736</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref10">
                <label>10</label>
                <mixed-citation>
                    <name>
                        <surname>Guadagna</surname>
                        <given-names>P.</given-names>
                    </name>
                    <name>
                        <surname>Fernandes</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Chen</surname>
                        <given-names>F.</given-names>
                    </name>
                    <name>
                        <surname>Santamaria</surname>
                        <given-names>A.</given-names>
                    </name>
                    <name>
                        <surname>Teng</surname>
                        <given-names>T.</given-names>
                    </name>
                    <name>
                        <surname>Frioni</surname>
                        <given-names>T.</given-names>
                    </name>
                    <name>
                        <surname>Caldwell</surname>
                        <given-names>D.</given-names>
                    </name>
                    <name>
                        <surname>Poni</surname>
                        <given-names>S.</given-names>
                    </name>
                    <name>
                        <surname>Semini</surname>
                        <given-names>C.</given-names>
                    </name>
                    <name>
                        <surname>Gatti</surname>
                        <given-names>M.</given-names>
                    </name>
                    <year>2023</year>
                    <article-title>Using deep learning for pruning region detection and plant organ segmentation in dormant spur-pruned grapevines</article-title>
                    <source>Precision Agriculture</source>
                    <volume>24</volume>
                    <issue>4</issue>
                    <page-range>1547-1569</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1007/s11119-023-10006-y">https://doi.org/10.1007/s11119-023-10006-y</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref11">
                <label>11</label>
                <mixed-citation>
                    <name>
                        <surname>Han</surname>
                        <given-names>K.</given-names>
                    </name>
                    <name>
                        <surname>Wang</surname>
                        <given-names>Y.</given-names>
                    </name>
                    <name>
                        <surname>Tian</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <name>
                        <surname>Guo</surname>
                        <given-names>J.</given-names>
                    </name>
                    <name>
                        <surname>Xu</surname>
                        <given-names>C.</given-names>
                    </name>
                    <name>
                        <surname>Xu</surname>
                        <given-names>C.</given-names>
                    </name>
                    <year>2020</year>
                    <article-title>GhostNet: More features from cheap operations</article-title>
                    <publisher-name>Paper presented at the 2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</publisher-name>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/CVPR42600.2020.00165">https://doi.org/10.1109/CVPR42600.2020.00165</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref12">
                <label>12</label>
                <mixed-citation>
                    <name>
                        <surname>Hou</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <name>
                        <surname>Zhou</surname>
                        <given-names>D.</given-names>
                    </name>
                    <name>
                        <surname>Feng</surname>
                        <given-names>J.</given-names>
                    </name>
                    <year>2021</year>
                    <article-title>Coordinate attention for efficient mobile network design</article-title>
                    <publisher-name>Paper presented at the 2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</publisher-name>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/CVPR46437.2021.01350">https://doi.org/10.1109/CVPR46437.2021.01350</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref13">
                <label>13</label>
                <mixed-citation>
                    <name>
                        <surname>Howard</surname>
                        <given-names>A.</given-names>
                    </name>
                    <name>
                        <surname>Sandler</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Chen</surname>
                        <given-names>B.</given-names>
                    </name>
                    <name>
                        <surname>Wang</surname>
                        <given-names>W.</given-names>
                    </name>
                    <name>
                        <surname>Chen</surname>
                        <given-names>L.</given-names>
                    </name>
                    <name>
                        <surname>Tan</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Chu</surname>
                        <given-names>G.</given-names>
                    </name>
                    <name>
                        <surname>Vasudevan</surname>
                        <given-names>V.</given-names>
                    </name>
                    <name>
                        <surname>Zhu</surname>
                        <given-names>Y.</given-names>
                    </name>
                    <name>
                        <surname>Pang</surname>
                        <given-names>R.</given-names>
                    </name>
                    <name>
                        <surname>Adam</surname>
                        <given-names>H.</given-names>
                    </name>
                    <name>
                        <surname>Le</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <year>2019</year>
                    <article-title>Searching for MobileNetV3</article-title>
                    <publisher-name>Paper presented at the 2019 IEEE/CVF International Conference on Computer Vision (ICCV)</publisher-name>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/ICCV.2019.00140">https://doi.org/10.1109/ICCV.2019.00140</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref14">
                <label>14</label>
                <mixed-citation>
                    <name>
                        <surname>Hu</surname>
                        <given-names>J.</given-names>
                    </name>
                    <name>
                        <surname>Shen</surname>
                        <given-names>L.</given-names>
                    </name>
                    <name>
                        <surname>Sun</surname>
                        <given-names>G.</given-names>
                    </name>
                    <year>2018</year>
                    <article-title>Squeeze-and-Excitation Networks</article-title>
                    <publisher-name>Paper presented at the 2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</publisher-name>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/CVPR.2018.00745">https://doi.org/10.1109/CVPR.2018.00745</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref15">
                <label>15</label>
                <mixed-citation>
                    <name>
                        <surname>Íñiguez</surname>
                        <given-names>R.</given-names>
                    </name>
                    <name>
                        <surname>Wolela</surname>
                        <given-names>F.</given-names>
                    </name>
                    <name>
                        <surname>Gonzalez Pavez</surname>
                        <given-names>M. I.</given-names>
                    </name>
                    <name>
                        <surname>Barrio</surname>
                        <given-names>I.</given-names>
                    </name>
                    <name>
                        <surname>Tardáguila</surname>
                        <given-names>J.</given-names>
                    </name>
                    <name>
                        <surname>Venter</surname>
                        <given-names>T.</given-names>
                    </name>
                    <name>
                        <surname>Poblete-Echeverria</surname>
                        <given-names>C.</given-names>
                    </name>
                    <year>2025</year>
                    <article-title>Artificial intelligence-driven classification method of grapevine major phenological stages using conventional RGB imaging</article-title>
                    <source>OENO One</source>
                    <volume>59</volume>
                    <issue>2</issue>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.20870/oeno-one.2025.59.2.9306">https://doi.org/10.20870/oeno-one.2025.59.2.9306</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref16">
                <label>16</label>
                <mixed-citation>
                    <name>
                        <surname>Karkee</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Majeed</surname>
                        <given-names>Y.</given-names>
                    </name>
                    <name>
                        <surname>Zhang</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <year>2023</year>
                    <article-title>Advanced technologies for crop-load management</article-title>
                    <source>In S. G. Vougioukas &amp; Q. Zhang (Eds.), Advanced Automation for Tree Fruit Orchards and Vineyards</source>
                    <comment>(p. 119-149). Cham: Springer International Publishing</comment>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1007/978-3-031-26941-7_6">https://doi.org/10.1007/978-3-031-26941-7_6</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref17">
                <label>17</label>
                <mixed-citation>
                    <name>
                        <surname>Ma</surname>
                        <given-names>N.</given-names>
                    </name>
                    <name>
                        <surname>Zhang</surname>
                        <given-names>X.</given-names>
                    </name>
                    <name>
                        <surname>Zheng</surname>
                        <given-names>H.</given-names>
                    </name>
                    <name>
                        <surname>Sun</surname>
                        <given-names>J.</given-names>
                    </name>
                    <year>2018</year>
                    <article-title>ShuffleNetV2: Practical guidelines for efficient CNN architecture design</article-title>
                    <source>Paper presented at the Computer Vision – ECCV 2018</source>
                    <volume>Cham</volume>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1007/978-3-030-01264-9_8">https://doi.org/10.1007/978-3-030-01264-9_8</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref18">
                <label>18</label>
                <mixed-citation>
                    <name>
                        <surname>Majeed</surname>
                        <given-names>Y.</given-names>
                    </name>
                    <name>
                        <surname>Karkee</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Zhang</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <year>2020b</year>
                    <article-title>Estimating the trajectories of vine cordons in full foliage canopies for automated green shoot thinning in vineyards</article-title>
                    <source>Computers and Electronics in Agriculture</source>
                    <volume>176</volume>
                    <page-range>105671</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1016/j.compag.2020.105671">https://doi.org/10.1016/j.compag.2020.105671</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref19">
                <label>19</label>
                <mixed-citation>
                    <name>
                        <surname>Majeed</surname>
                        <given-names>Y.</given-names>
                    </name>
                    <name>
                        <surname>Karkee</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Zhang</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <name>
                        <surname>Fu</surname>
                        <given-names>L.</given-names>
                    </name>
                    <name>
                        <surname>Whiting</surname>
                        <given-names>M. D.</given-names>
                    </name>
                    <year>2019</year>
                    <article-title>A study on the detection of visible parts of cordons using deep learning networks for automated green shoot thinning in vineyards</article-title>
                    <source>IFAC-PapersOnLine</source>
                    <volume>52</volume>
                    <issue>30</issue>
                    <page-range>82-86</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1016/j.ifacol.2019.12.501">https://doi.org/10.1016/j.ifacol.2019.12.501</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref20">
                <label>20</label>
                <mixed-citation>
                    <name>
                        <surname>Majeed</surname>
                        <given-names>Y.</given-names>
                    </name>
                    <name>
                        <surname>Karkee</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Zhang</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <name>
                        <surname>Fu</surname>
                        <given-names>L.</given-names>
                    </name>
                    <name>
                        <surname>Whiting</surname>
                        <given-names>M. D.</given-names>
                    </name>
                    <year>2020a</year>
                    <article-title>Determining grapevine cordon shape for automated green shoot thinning using semantic segmentation-based deep learning networks</article-title>
                    <source>Computers and Electronics in Agriculture</source>
                    <volume>171</volume>
                    <page-range>105308</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1016/j.compag.2020.105308">https://doi.org/10.1016/j.compag.2020.105308</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref21">
                <label>21</label>
                <mixed-citation>
                    <name>
                        <surname>Marset</surname>
                        <given-names>W. V.</given-names>
                    </name>
                    <name>
                        <surname>Pérez</surname>
                        <given-names>D. S.</given-names>
                    </name>
                    <name>
                        <surname>Díaz</surname>
                        <given-names>C. A.</given-names>
                    </name>
                    <name>
                        <surname>Bromberg</surname>
                        <given-names>F.</given-names>
                    </name>
                    <year>2021</year>
                    <article-title>Towards practical 2D grapevine bud detection with fully convolutional networks</article-title>
                    <source>Computers and Electronics in Agriculture</source>
                    <volume>182</volume>
                    <page-range>105947</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1016/j.compag.2020.105947">https://doi.org/10.1016/j.compag.2020.105947</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref22">
                <label>22</label>
                <mixed-citation>
                    <name>
                        <surname>Mcfarlane</surname>
                        <given-names>N. J. B.</given-names>
                    </name>
                    <name>
                        <surname>Tisseyre</surname>
                        <given-names>B.</given-names>
                    </name>
                    <name>
                        <surname>Sinfort</surname>
                        <given-names>C.</given-names>
                    </name>
                    <name>
                        <surname>Tillett</surname>
                        <given-names>R. D.</given-names>
                    </name>
                    <name>
                        <surname>Sevila</surname>
                        <given-names>F.</given-names>
                    </name>
                    <year>1997</year>
                    <article-title>Image analysis for pruning of long wood grape vines</article-title>
                    <source>Journal of Agricultural Engineering Research</source>
                    <volume>66</volume>
                    <issue>2</issue>
                    <page-range>111-119</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1006/jaer.1996.0125">https://doi.org/10.1006/jaer.1996.0125</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref23">
                <label>23</label>
                <mixed-citation>
                    <name>
                        <surname>Mercurio</surname>
                        <given-names>J. F.</given-names>
                    </name>
                    <name>
                        <surname>Gunkel</surname>
                        <given-names>W. W.</given-names>
                    </name>
                    <name>
                        <surname>Sobel</surname>
                        <given-names>T. A.</given-names>
                    </name>
                    <name>
                        <surname>Throop</surname>
                        <given-names>J. A.</given-names>
                    </name>
                    <name>
                        <surname>Norman</surname>
                        <given-names>D. W.</given-names>
                    </name>
                    <year>1989</year>
                    <article-title>Vision-guided block-type robotic grapevine pruner</article-title>
                    <source>ASAE paper no. 89–7519</source>
                    <volume>New Orleans</volume>
                    <comment>USA, 12–15 Dec.</comment>
                </mixed-citation>
            </ref>
            <ref id="ref24">
                <label>24</label>
                <mixed-citation>
                    <name>
                        <surname>Moreno</surname>
                        <given-names>H.</given-names>
                    </name>
                    <name>
                        <surname>Andújar</surname>
                        <given-names>D.</given-names>
                    </name>
                    <year>2023</year>
                    <article-title>Proximal sensing for geometric characterization of vines: A review of the latest advances</article-title>
                    <source>Computers and Electronics in Agriculture</source>
                    <volume>210</volume>
                    <page-range>107901</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1016/j.compag.2023.107901">https://doi.org/10.1016/j.compag.2023.107901</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref25">
                <label>25</label>
                <mixed-citation>
                    <name>
                        <surname>Ouyang</surname>
                        <given-names>D.</given-names>
                    </name>
                    <name>
                        <surname>He</surname>
                        <given-names>S.</given-names>
                    </name>
                    <name>
                        <surname>Zhang</surname>
                        <given-names>G.</given-names>
                    </name>
                    <name>
                        <surname>Luo</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Guo</surname>
                        <given-names>H.</given-names>
                    </name>
                    <name>
                        <surname>Zhan</surname>
                        <given-names>J.</given-names>
                    </name>
                    <name>
                        <surname>Huang</surname>
                        <given-names>Z.</given-names>
                    </name>
                    <year>2023</year>
                    <article-title>Efficient multi-scale attention module with cross-spatial learning</article-title>
                    <source>Paper presented at the ICASSP 2023 – 2023 IEEE International Conference on Acoustics</source>
                    <volume>Speech and Signal Processing</volume>
                    <issue>ICASSP</issue>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/ICASSP49357.2023.10096516">https://doi.org/10.1109/ICASSP49357.2023.10096516</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref26">
                <label>26</label>
                <mixed-citation>
                    <name>
                        <surname>Rana</surname>
                        <given-names>S.</given-names>
                    </name>
                    <name>
                        <surname>Gerbino</surname>
                        <given-names>S.</given-names>
                    </name>
                    <name>
                        <surname>Akbari </surname>
                        <given-names>sekehravani</given-names>
                    </name>
                    <name>
                        <surname>E</surname>
                        <given-names/>
                    </name>
                    <name>
                        <surname>Russo</surname>
                        <given-names>M. B.</given-names>
                    </name>
                    <name>
                        <surname>Carillo</surname>
                        <given-names>P.</given-names>
                    </name>
                    <year>2024</year>
                    <article-title>Crop growth analysis using automatic annotations and transfer learning in multi-date aerial images and ortho-mosaics</article-title>
                    <source>Agronomy</source>
                    <volume>14</volume>
                    <issue>9</issue>
                    <page-range>2052</page-range>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.3390/agronomy14092052">https://doi.org/10.3390/agronomy14092052</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref27">
                <label>27</label>
                <mixed-citation>
                    <name>
                        <surname>Sandler</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Howard</surname>
                        <given-names>A.</given-names>
                    </name>
                    <name>
                        <surname>Zhu</surname>
                        <given-names>M.</given-names>
                    </name>
                    <name>
                        <surname>Zhmoginov</surname>
                        <given-names>A.</given-names>
                    </name>
                    <name>
                        <surname>Chen</surname>
                        <given-names>L. C.</given-names>
                    </name>
                    <year>2018</year>
                    <article-title>MobileNetV2: Inverted residuals and linear bottlenecks</article-title>
                    <publisher-name>Paper presented at the 2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</publisher-name>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/CVPR.2018.00474">https://doi.org/10.1109/CVPR.2018.00474</ext-link>
                </mixed-citation>
            </ref>
            <ref id="ref28">
                <label>28</label>
                <mixed-citation>
                    <name>
                        <surname>Wang</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <name>
                        <surname>Wu</surname>
                        <given-names>B.</given-names>
                    </name>
                    <name>
                        <surname>Zhu</surname>
                        <given-names>P.</given-names>
                    </name>
                    <name>
                        <surname>Li</surname>
                        <given-names>P.</given-names>
                    </name>
                    <name>
                        <surname>Zuo</surname>
                        <given-names>W.</given-names>
                    </name>
                    <name>
                        <surname>Hu</surname>
                        <given-names>Q.</given-names>
                    </name>
                    <year>2020</year>
                    <article-title>ECA-Net: Efficient channel attention for deep convolutional neural networks</article-title>
                    <publisher-name>Paper presented at the 2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</publisher-name>
                    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/CVPR42600.2020.01155">https://doi.org/10.1109/CVPR42600.2020.01155</ext-link>
                </mixed-citation>
            </ref>
        </ref-list>
    </back>
</article>
